smart-data-engine-sdk 0.1.0__py3-none-any.whl

This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
Files changed (73) hide show
  1. sde/__init__.py +318 -0
  2. sde/_cutover_project.py +179 -0
  3. sde/_local_state.py +188 -0
  4. sde/_operator_deadline.py +50 -0
  5. sde/_usage.py +314 -0
  6. sde/bulk.py +79 -0
  7. sde/canonical.py +141 -0
  8. sde/capabilities.py +62 -0
  9. sde/cutover.py +286 -0
  10. sde/engines/__init__.py +0 -0
  11. sde/engines/_clickhouse_connection.py +224 -0
  12. sde/engines/_index_build.py +294 -0
  13. sde/engines/_operator.py +394 -0
  14. sde/engines/_staging.py +222 -0
  15. sde/engines/_storage.py +22 -0
  16. sde/engines/_write_fences.py +271 -0
  17. sde/engines/clickhouse.py +1115 -0
  18. sde/engines/orderbook.py +457 -0
  19. sde/engines/postgres.py +967 -0
  20. sde/entity.py +170 -0
  21. sde/errors.py +103 -0
  22. sde/explain.py +300 -0
  23. sde/frozen_verification.py +152 -0
  24. sde/generation.py +131 -0
  25. sde/groups.py +97 -0
  26. sde/hashing.py +242 -0
  27. sde/index_build.py +313 -0
  28. sde/index_operator.py +347 -0
  29. sde/infer.py +461 -0
  30. sde/inspection.py +62 -0
  31. sde/internal.py +90 -0
  32. sde/layout.py +669 -0
  33. sde/local_cutover.py +801 -0
  34. sde/logging.py +143 -0
  35. sde/migration.py +856 -0
  36. sde/model.py +482 -0
  37. sde/physical.py +531 -0
  38. sde/placement.py +1010 -0
  39. sde/provisioning.py +63 -0
  40. sde/py.typed +0 -0
  41. sde/query.py +521 -0
  42. sde/routing.py +85 -0
  43. sde/schema.py +466 -0
  44. sde/session.py +993 -0
  45. sde/shapes.py +153 -0
  46. sde/staging.py +264 -0
  47. sde/staging_operator.py +393 -0
  48. sde/telemetry.py +1087 -0
  49. sde/testing/__init__.py +14 -0
  50. sde/testing/loader.py +175 -0
  51. sde/testing/memory.py +331 -0
  52. sde/types.py +228 -0
  53. sde/verification.py +220 -0
  54. sde/watermark.py +222 -0
  55. sde/write_fence.py +283 -0
  56. sde_demo/__init__.py +1 -0
  57. sde_demo/__main__.py +183 -0
  58. sde_demo/diagnostics.py +92 -0
  59. sde_demo/model.py +75 -0
  60. sde_demo/project.py +312 -0
  61. sde_demo/py.typed +0 -0
  62. sde_demo/query_count.py +301 -0
  63. sde_demo/resources.py +969 -0
  64. sde_demo/runtime.py +419 -0
  65. sde_demo/verification.py +242 -0
  66. sde_operator/__init__.py +1 -0
  67. sde_operator/__main__.py +210 -0
  68. smart_data_engine_sdk-0.1.0.dist-info/METADATA +174 -0
  69. smart_data_engine_sdk-0.1.0.dist-info/RECORD +73 -0
  70. smart_data_engine_sdk-0.1.0.dist-info/WHEEL +4 -0
  71. smart_data_engine_sdk-0.1.0.dist-info/entry_points.txt +3 -0
  72. smart_data_engine_sdk-0.1.0.dist-info/licenses/LICENSE +201 -0
  73. smart_data_engine_sdk-0.1.0.dist-info/licenses/NOTICE +13 -0
@@ -0,0 +1,1115 @@
1
+ """ClickHouse adapter.
2
+
3
+ The second engine, and the reason there is a product at all: between PostgreSQL and ClickHouse
4
+ lies the decision clients get wrong once, at the start, and never revisit. With one adapter the
5
+ planner chooses from a set of size one.
6
+
7
+ Three things here are genuinely different from the PostgreSQL adapter, and none of them is a
8
+ detail. Each is a place where "the same operation" means something else because the engine is
9
+ different, and the product's position is that such a difference must be *stated* rather than
10
+ smoothed over.
11
+
12
+ **A naive datetime is UTC. Always, explicitly, here.**
13
+
14
+ A fix for a divergence that was measured rather than imagined. `datetime(2026, 8, 27, 12, 0)` with
15
+ no tzinfo, written through this library:
16
+
17
+ - into PostgreSQL `timestamptz`, it arrives as ``12:00:00+00:00`` - into ClickHouse
18
+ ``DateTime64(3, 'UTC')`` with the driver left to its own devices, it arrived as
19
+ ``10:00:00`` - because `clickhouse-connect` reads a naive datetime as *local* time and converts
20
+
21
+ Two hours apart, from one call, in one library, with no error anywhere. After a migration from one
22
+ engine to the other every timestamp in the client's analytics would shift by the offset of
23
+ whichever machine happened to write the row. So naive datetimes are given `timezone.utc` before
24
+ they reach the driver, which is what the PostgreSQL path already does in effect. If you mean a
25
+ different zone, pass an aware datetime and it is respected.
26
+
27
+ **There are no transactions, and this adapter says so instead of pretending.**
28
+
29
+ `transaction()` raises. Not a no-op context manager: a caller who believes they are in a
30
+ transaction and is not has been lied to at the worst possible moment. A client who needs two
31
+ entities to commit together declares that, and the planner then places them in the same group and
32
+ therefore the same engine - the requirement becomes a placement constraint. This method existing
33
+ and refusing is how that constraint is discovered at the point of use rather than inferred from a
34
+ rollback that never happened.
35
+
36
+ **Keys are not enforced, so the table is a `ReplacingMergeTree` and point reads use `FINAL`.**
37
+
38
+ A MergeTree does not enforce uniqueness on its `ORDER BY`. So a second `save()` of the same key
39
+ does not raise the way PostgreSQL's primary key does - and that divergence cannot be removed, only
40
+ chosen. Plain `MergeTree` would leave two rows and make every aggregate over that group quietly
41
+ wrong; `ReplacingMergeTree` keeps the newest and collapses the rest at merge time. Of the two,
42
+ "the newest row for a key wins" is something a client can reason about. `FINAL` on point reads and
43
+ counts is what makes the collapse visible immediately rather than eventually, and it is not free -
44
+ which is the trade, stated.
45
+
46
+ The residue is worth naming plainly, because a client will meet it: in PostgreSQL, saving the same
47
+ key twice is an error; here it is an overwrite. That belongs in a placement decision rather than
48
+ in an adapter, and there is no declaration for it yet.
49
+ """
50
+
51
+ from __future__ import annotations
52
+
53
+ import datetime as _dt
54
+ import math
55
+ import os
56
+ import re
57
+ import ssl
58
+ import stat
59
+ from collections.abc import Iterator, Mapping, Sequence
60
+ from importlib.metadata import version
61
+ from typing import Any, cast
62
+
63
+ from .._usage import UsageGate, guarded
64
+ from ..bulk import batch_columns
65
+ from ..errors import EngineError
66
+ from ..explain import (
67
+ Cost,
68
+ QueryPlan,
69
+ QueryPlanRefused,
70
+ replacing_merge_tree_finding,
71
+ )
72
+ from ..logging import log
73
+ from ..migration import key_columns, same_width
74
+ from ..physical import (
75
+ PhysicalFinding,
76
+ declared_tables,
77
+ parse_identifier_list,
78
+ parse_partition_key,
79
+ )
80
+ from ..placement import BACKFILL_TABLE, WATERMARK_TABLE, PhysicalLayout
81
+ from ..query import ReadColumn, ReadPlan, read_row, read_sql, summary_sql
82
+ from ..schema import QUOTE, schema_statements
83
+ from ..write_fence import WriteFence
84
+ from ._clickhouse_connection import (
85
+ CONNECT_TIMEOUT_SECONDS as CONNECT_TIMEOUT_SECONDS,
86
+ )
87
+ from ._clickhouse_connection import (
88
+ HANDSHAKE_TIMEOUT_SECONDS as HANDSHAKE_TIMEOUT_SECONDS,
89
+ )
90
+ from ._clickhouse_connection import (
91
+ ConnectionParameters,
92
+ parse_dsn,
93
+ )
94
+ from ._storage import STORAGE_COLUMNS
95
+ from ._write_fences import ClickHouseFences
96
+
97
+ __all__ = ["STORAGE_COLUMNS", "ClickHouseEngine"]
98
+
99
+
100
+ # Bound from the one definition in sde.schema, so that DDL and DML cannot disagree about
101
+ # how an identifier is escaped.
102
+ _quote = QUOTE["clickhouse"]
103
+
104
+
105
+ def _as_utc(value: Any) -> Any:
106
+ """A naive datetime is UTC. See the module docstring for the measurement behind this."""
107
+ if isinstance(value, _dt.datetime) and value.tzinfo is None:
108
+ return value.replace(tzinfo=_dt.UTC)
109
+ return value
110
+
111
+
112
+ def _query_parameter(name: str, value: Any, parameters: dict[str, Any]) -> str:
113
+ """Bind a scalar without letting the driver's datetime formatter discard its fraction.
114
+
115
+ The plain datetime formatter emits seconds. Use a bound string and an explicit UTC
116
+ DateTime64 conversion for temporal comparisons, independent of the driver's formatting zone.
117
+ Data values still go through the driver's parameter binding; none are interpolated into SQL.
118
+ """
119
+ placeholder = f"%({name})s"
120
+ if isinstance(value, _dt.datetime):
121
+ utc = _as_utc(value).astimezone(_dt.UTC)
122
+ parameters[name] = utc.replace(tzinfo=None).isoformat(" ", timespec="microseconds")
123
+ return f"toDateTime64({placeholder}, 6, 'UTC')"
124
+ parameters[name] = value
125
+ return placeholder
126
+
127
+
128
+ def _row(names: Sequence[str], types: Sequence[Any], row: Sequence[Any]) -> dict[str, Any]:
129
+ """One result row as a dict, with the timezone put back on values that had one in the schema.
130
+
131
+ The write side of this module makes a naive datetime UTC. Without this, the read side would
132
+ not give it back: `clickhouse-connect` returns a *naive* datetime for a `DateTime64(3, 'UTC')`
133
+ column, while psycopg returns an aware one for a `timestamptz`. So the same entity read from
134
+ the two engines produces two datetimes that Python refuses to compare - ``can't compare
135
+ offset-naive and offset-aware datetimes`` - which is a `TypeError` in a client's code that
136
+ appears on the day a group is moved and not before.
137
+
138
+ The column type carries the zone, so it is read from there rather than assumed: a column
139
+ declared with a timezone yields aware values, one declared without yields naive, which is
140
+ exactly the distinction the neutral vocabulary makes between `timestamptz` and `timestamp`.
141
+ """
142
+ out: dict[str, Any] = {}
143
+ for name, column_type, value in zip(names, types, row, strict=True):
144
+ zone = getattr(column_type, "tzinfo", None)
145
+ if zone is not None and isinstance(value, _dt.datetime) and value.tzinfo is None:
146
+ value = value.replace(tzinfo=zone)
147
+ out[name] = value
148
+ return out
149
+
150
+
151
+ class _NoReplayTransport:
152
+ """Keep the driver's TLS/proxy pool but forbid ambiguous transport replay for this client.
153
+
154
+ clickhouse-connect retries RemoteDisconnected even when its retry count is zero. In a live
155
+ probe that error followed an accepted INSERT: the driver replayed it and returned success.
156
+ Translate transport errors before that retry handler, and disable urllib3's own retry layer.
157
+ The shared pool is never patched; only this client's reference is wrapped.
158
+ """
159
+
160
+ def __init__(self, pool: Any) -> None:
161
+ self.pool = pool
162
+
163
+ def __getattr__(self, name: str) -> Any:
164
+ return getattr(self.pool, name)
165
+
166
+ # The driver keys pool-expiration bookkeeping by manager. Preserve that identity instead of
167
+ # registering a new shared-pool owner on every connection (and clearing its active sockets).
168
+ def __hash__(self) -> int:
169
+ return hash(self.pool)
170
+
171
+ def __eq__(self, other: object) -> bool:
172
+ return self.pool is (other.pool if isinstance(other, _NoReplayTransport) else other)
173
+
174
+ def request(self, method: str, url: str, **options: Any) -> Any:
175
+ from urllib3.exceptions import HTTPError
176
+
177
+ options["retries"] = False
178
+ options["redirect"] = False
179
+ try:
180
+ return self.pool.request(method, url, **options)
181
+ except HTTPError as exc:
182
+ raise EngineError(
183
+ "ClickHouse transport failed; the operation was not replayed and its outcome "
184
+ "may be unknown"
185
+ ) from exc
186
+
187
+
188
+ MIN_CLICKHOUSE_CONNECT = "1.7.2"
189
+
190
+
191
+ def _require_driver_version(installed: str) -> None:
192
+ parts = re.fullmatch(r"(\d+)\.(\d+)\.(\d+)(?:\.post\d+)?", installed)
193
+ if parts is None or tuple(int(part) for part in parts.groups()) < tuple(
194
+ int(part) for part in MIN_CLICKHOUSE_CONNECT.split(".")
195
+ ):
196
+ raise EngineError(
197
+ f"ClickHouse requires stable clickhouse-connect >= {MIN_CLICKHOUSE_CONNECT}; "
198
+ f"installed {installed}. Older drivers do not meet the timestamp and structured "
199
+ "error-code contract. Install the SDK clickhouse extra in this environment."
200
+ )
201
+
202
+
203
+ MAX_CA_BYTES = 1024 * 1024
204
+
205
+
206
+ def _ca_context(path: str | None) -> ssl.SSLContext | None:
207
+ """Read and qualify private per-client trust before creating any driver connection."""
208
+ if path is None:
209
+ return None
210
+ try:
211
+ flags = os.O_RDONLY | getattr(os, "O_NONBLOCK", 0) | getattr(os, "O_CLOEXEC", 0)
212
+ fd = os.open(path, flags)
213
+ try:
214
+ info = os.fstat(fd)
215
+ if not stat.S_ISREG(info.st_mode) or not 0 < info.st_size <= MAX_CA_BYTES:
216
+ raise ValueError("CA must be a bounded regular file")
217
+ pieces: list[bytes] = []
218
+ size = 0
219
+ while size <= MAX_CA_BYTES:
220
+ piece = os.read(fd, min(65536, MAX_CA_BYTES + 1 - size))
221
+ if not piece:
222
+ break
223
+ pieces.append(piece)
224
+ size += len(piece)
225
+ payload = b"".join(pieces)
226
+ finally:
227
+ os.close(fd)
228
+ if not payload or len(payload) > MAX_CA_BYTES:
229
+ raise ValueError("CA must be nonempty and at most 1 MiB")
230
+ context = ssl.SSLContext(ssl.PROTOCOL_TLS_CLIENT)
231
+ context.load_verify_locations(cadata=payload.decode("ascii"))
232
+ if not context.cert_store_stats()["x509"]:
233
+ raise ValueError("empty CA store")
234
+ return context
235
+ except (OSError, ValueError, UnicodeError):
236
+ raise EngineError(
237
+ "ClickHouse CA file must contain readable valid PEM certificates within 1 MiB"
238
+ ) from None
239
+
240
+
241
+ def _connect_client(parameters: ConnectionParameters) -> Any:
242
+ from clickhouse_connect.driver.httpclient import HttpClient
243
+ from clickhouse_connect.driver.query import TzSource
244
+ from urllib3 import Timeout
245
+
246
+ context = _ca_context(parameters.ca_cert)
247
+
248
+ class GuardedHttpClient(HttpClient):
249
+ def _init_common_settings(self, tz_source: TzSource) -> None:
250
+ # HttpClient calls this after constructing its native pool/backend and before the
251
+ # initial version/settings exchange. Change only this client's pool reference.
252
+ try:
253
+ self.http = cast(Any, _NoReplayTransport(self.http))
254
+ self.timeout = Timeout(
255
+ connect=parameters.connect_timeout, read=parameters.send_receive_timeout
256
+ )
257
+ if context is not None:
258
+ if not self._owns_pool_manager:
259
+ raise EngineError("The driver did not create an isolated CA pool")
260
+ # The native factory owns this pool; pin the already validated CA bytes so
261
+ # later sockets do not reread a changed path. Never alter the shared pool.
262
+ pool_options = self.http.connection_pool_kw
263
+ pool_options["ssl_context"] = context
264
+ pool_options.pop("ca_certs", None)
265
+ super()._init_common_settings(tz_source)
266
+ except BaseException as exc:
267
+ try:
268
+ self.close()
269
+ except BaseException:
270
+ exc.add_note("ClickHouse client cleanup also failed after startup.")
271
+ raise
272
+
273
+ # Driver 1.7.2 coerces timeout inputs to integers before constructing its backend. Positive
274
+ # placeholders permit construction; the scoped hook installs the exact requested fractions
275
+ # before I/O. The URI itself is never reparsed or used as an arbitrary kwargs channel.
276
+ host = f"[{parameters.host}]" if ":" in parameters.host else parameters.host
277
+ return GuardedHttpClient(
278
+ interface=parameters.interface,
279
+ host=host,
280
+ port=parameters.port,
281
+ username=parameters.username,
282
+ password=parameters.password,
283
+ database=parameters.database,
284
+ verify=True,
285
+ ca_cert=parameters.ca_cert,
286
+ query_retries=0,
287
+ connect_timeout=math.ceil(parameters.connect_timeout),
288
+ send_receive_timeout=math.ceil(parameters.send_receive_timeout),
289
+ )
290
+
291
+
292
+ class ClickHouseEngine:
293
+ """A thin adapter over clickhouse-connect. Executes decisions, makes none."""
294
+
295
+ dialect = "clickhouse"
296
+
297
+ def __init__(self, dsn: str) -> None:
298
+ self._connection = parse_dsn(dsn)
299
+ try:
300
+ import clickhouse_connect # noqa: F401 -- validate the optional driver extra
301
+ except ImportError as exc: # pragma: no cover - depends on the install extra
302
+ raise EngineError(
303
+ "the ClickHouse adapter needs the 'clickhouse' extra: "
304
+ "pip install 'smart-data-engine-sdk[clickhouse]'. "
305
+ "The core library has no dependencies, because it goes into your application and "
306
+ "every dependency here would be one you inherit."
307
+ ) from exc
308
+ _require_driver_version(version("clickhouse-connect"))
309
+ self._dsn = dsn
310
+ self._client: Any = None
311
+ self._usage = UsageGate()
312
+
313
+ # --- connection ------------------------------------------------------------------------
314
+
315
+ @guarded
316
+ def connect(self) -> None:
317
+ """Open the client, with a bound on how long that may take.
318
+
319
+ Same measurement as the PostgreSQL adapter, and the same absence: a host that accepts the
320
+ connection and never answers hung for as long as the test would wait. Here the reason is
321
+ one layer up - the TCP connect succeeds, so ``connect_timeout`` never fires, and what is
322
+ left is ``send_receive_timeout``, whose driver default is 300 seconds.
323
+
324
+ Both are transport timers, with distinct defaults for opening and the initial exchange.
325
+ Exact caller-supplied seconds survive the driver's integer constructor coercion. Socket
326
+ inactivity is not an absolute query deadline: progress received from a long-running query
327
+ can keep that exchange alive under the driver's existing transport behavior.
328
+ """
329
+ if self._client is None:
330
+ try:
331
+ self._client = _connect_client(self._connection)
332
+ except EngineError as exc:
333
+ raise EngineError("could not connect to ClickHouse: " + str(exc)) from None
334
+ except Exception:
335
+ # Driver/parser exceptions may echo URL, credentials or server content. The
336
+ # connection error conveys no proof of rollback and never weakens TLS to retry.
337
+ raise EngineError(
338
+ "could not connect to ClickHouse with the configured transport"
339
+ ) from None
340
+
341
+ @guarded
342
+ def close(self) -> None:
343
+ if self._client is not None:
344
+ self._client.close()
345
+ self._client = None
346
+
347
+ def __enter__(self) -> ClickHouseEngine:
348
+ self.connect()
349
+ return self
350
+
351
+ def __exit__(self, *_: object) -> None:
352
+ self.close()
353
+
354
+ @property
355
+ def _cx(self) -> Any:
356
+ if self._client is None:
357
+ raise EngineError("not connected; call connect() first")
358
+ return self._client
359
+
360
+ def write_fence(self, table: str, *, project_id: str) -> WriteFence:
361
+ """DDL capability; use a dedicated connection separate from application traffic."""
362
+ return WriteFence(ClickHouseFences(self._cx), table, project_id=project_id)
363
+
364
+ # --- schema ----------------------------------------------------------------------------
365
+
366
+ @guarded
367
+ def ensure_schema(
368
+ self, layout: PhysicalLayout, *, keys: Mapping[str, Sequence[str]]
369
+ ) -> tuple[PhysicalFinding, ...]:
370
+ """Create what is missing, change nothing that exists.
371
+
372
+ `ORDER BY` is the declared key, in declared order. That order is positional and carries
373
+ meaning: it decides which prefixes of the key can prune granules, so sorting it would
374
+ change the physical performance of the table while leaving the map looking identical.
375
+
376
+ Data-skipping indexes, when a layout declares them (map contract 5), are part of the
377
+ `CREATE TABLE` statement: `IF NOT EXISTS` never adds one to a table that already exists.
378
+
379
+ Returns how existing tables differ from the declared physical design - sort key, partition,
380
+ indexes - without refusing: the caller decides whether a difference is a refusal (a person
381
+ provisioning a map) or a report (a running session). Columns and types still refuse here.
382
+ """
383
+ statements = schema_statements(layout, keys=keys, dialect=self.dialect)
384
+
385
+ for statement in statements:
386
+ try:
387
+ self._cx.command(statement)
388
+ except Exception as exc:
389
+ raise EngineError(f"schema statement failed: {statement}: {exc}") from exc
390
+ log("sde.schema.applied", engine=self.dialect, statements=len(statements))
391
+ self._verify_schema(layout)
392
+ return self._physical_findings(layout, keys)
393
+
394
+ @guarded
395
+ def validate_schema(
396
+ self, layout: PhysicalLayout, *, keys: Mapping[str, Sequence[str]] | None = None
397
+ ) -> tuple[PhysicalFinding, ...]:
398
+ """Check the existing physical columns without issuing DDL; report the physical design."""
399
+ self._verify_schema(layout)
400
+ return () if keys is None else self._physical_findings(layout, keys)
401
+
402
+ def _physical_findings(
403
+ self, layout: PhysicalLayout, keys: Mapping[str, Sequence[str]]
404
+ ) -> tuple[PhysicalFinding, ...]:
405
+ """Sort key, partition and data-skipping indexes as the catalogue reports them.
406
+
407
+ The catalogue formats expressions, so they are parsed into names rather than compared as
408
+ strings we predict (:func:`sde.physical.parse_identifier_list`). A key or partition the
409
+ parser does not understand is reported as it stands - an expression we cannot read is not
410
+ evidence that the table matches.
411
+ """
412
+ declared = declared_tables(layout, keys)
413
+ if not declared:
414
+ return ()
415
+ tables = [entry.table for entry in declared]
416
+ described = self._cx.query(
417
+ "SELECT name, sorting_key, partition_key FROM system.tables "
418
+ "WHERE database = currentDatabase() AND name IN %(tables)s",
419
+ parameters={"tables": tables},
420
+ ).result_rows
421
+ keys_found = {str(name): (str(sort), str(part)) for name, sort, part in described}
422
+ present: dict[str, dict[str, tuple[str, str, int]]] = {}
423
+ unreadable = ""
424
+ indexed = sorted(entry.table for entry in declared if entry.indexes)
425
+ # Read only when something is declared. `system.tables` and `system.columns` are filtered
426
+ # by the login's own table privileges, so a restricted runtime login reads them without a
427
+ # grant; `system.data_skipping_indices` needs an explicit one (measured, 24.8: code 497).
428
+ # Reading it for every table would have made every session of an existing deployment
429
+ # fail to start after an upgrade, for a check about performance.
430
+ if indexed:
431
+ try:
432
+ index_rows = self._cx.query(
433
+ "SELECT table, name, type_full, expr, granularity "
434
+ "FROM system.data_skipping_indices "
435
+ "WHERE database = currentDatabase() AND table IN %(tables)s",
436
+ parameters={"tables": indexed},
437
+ ).result_rows
438
+ except Exception as exc:
439
+ if getattr(exc, "code", None) != 497:
440
+ raise
441
+ # Unverified, not matching: a running session reports this and goes on; a person
442
+ # provisioning refuses on it, because "could not look" is not "looked and agreed".
443
+ unreadable = (
444
+ "unverified: this login cannot read system.data_skipping_indices "
445
+ "(GRANT SELECT ON system.data_skipping_indices to verify it)"
446
+ )
447
+ index_rows = []
448
+ for table_name, index_name, type_full, expr, granularity in index_rows:
449
+ present.setdefault(str(table_name), {})[str(index_name)] = (
450
+ str(type_full),
451
+ str(expr),
452
+ int(granularity),
453
+ )
454
+ findings: list[PhysicalFinding] = []
455
+ for entry in declared:
456
+ if entry.table not in keys_found:
457
+ continue # _verify_schema has already refused a missing table
458
+ sorting, partition = keys_found[entry.table]
459
+ try:
460
+ found_key: tuple[str, ...] | str = parse_identifier_list(sorting)
461
+ except ValueError:
462
+ found_key = repr(sorting)
463
+ if found_key != entry.key:
464
+ findings.append(
465
+ PhysicalFinding(entry.table, "sort key", repr(list(entry.key)), str(found_key))
466
+ )
467
+ try:
468
+ found_partition: tuple[str, str] | str | None = parse_partition_key(partition)
469
+ except ValueError:
470
+ found_partition = repr(partition)
471
+ if found_partition != entry.partition:
472
+ findings.append(
473
+ PhysicalFinding(
474
+ entry.table, "partition", str(entry.partition), str(found_partition)
475
+ )
476
+ )
477
+ for index in entry.indexes:
478
+ got = present.get(entry.table, {}).get(index.name)
479
+ wanted = (
480
+ f"{index.type_full} on {list(index.columns)} granularity {index.granularity}"
481
+ )
482
+ if got is None:
483
+ findings.append(
484
+ PhysicalFinding(
485
+ entry.table, f"index {index.name}", wanted, unreadable or "absent"
486
+ )
487
+ )
488
+ continue
489
+ type_full, expr, granularity = got
490
+ try:
491
+ columns: tuple[str, ...] | str = parse_identifier_list(expr)
492
+ except ValueError:
493
+ columns = repr(expr)
494
+ if (
495
+ type_full != index.type_full
496
+ or columns != index.columns
497
+ or granularity != index.granularity
498
+ ):
499
+ findings.append(
500
+ PhysicalFinding(
501
+ entry.table,
502
+ f"index {index.name}",
503
+ wanted,
504
+ f"{type_full} on {columns} granularity {granularity}",
505
+ )
506
+ )
507
+ return tuple(findings)
508
+
509
+ def _verify_schema(self, layout: PhysicalLayout) -> None:
510
+ """The same check as the PostgreSQL adapter, for the same reason.
511
+
512
+ `CREATE TABLE IF NOT EXISTS` keeps whatever table is already there, so a leftover from an
513
+ older map is accepted in silence and the first insert fails naming a column rather than
514
+ the cause. Missing columns are refused, extra ones are logged: a client may have added one
515
+ outside SDE and the map has no opinion about it.
516
+
517
+ **Types too, and here they need no normalising at all.** Measured against every type this
518
+ library renders: `system.columns.type` returns the exact string we wrote, down to the space
519
+ in `Decimal(12, 2)` and the quotes in `DateTime64(6, 'UTC')`. So the comparison is literal,
520
+ and it is literal on purpose - a renderer that started emitting a different spelling of the
521
+ same type would fail this check, which is the right way round for a document we sign.
522
+ """
523
+ expected = {
524
+ table: dict(layout.columns.get(entity, {}))
525
+ for entity, table in sorted(layout.tables.items())
526
+ }
527
+ if not expected:
528
+ return
529
+
530
+ result = self._cx.query(
531
+ "SELECT table, name, type FROM system.columns "
532
+ "WHERE database = currentDatabase() AND table IN %(tables)s",
533
+ parameters={"tables": sorted(expected)},
534
+ )
535
+ found: dict[str, dict[str, str]] = {}
536
+ for table_name, column_name, column_type in result.result_rows:
537
+ found.setdefault(str(table_name), {})[str(column_name)] = str(column_type)
538
+
539
+ for table, columns in sorted(expected.items()):
540
+ actual = found.get(table)
541
+ if actual is None:
542
+ raise EngineError(
543
+ f"{table!r} does not exist after applying the schema. The statement reported "
544
+ f"success, so this is a permissions or database-selection problem rather "
545
+ f"than a bad map."
546
+ )
547
+ missing = sorted(set(columns) - set(actual))
548
+ if missing:
549
+ raise EngineError(
550
+ f"{table!r} already existed with a different shape: the map needs "
551
+ f"{missing} and the table has {sorted(actual)}. `CREATE TABLE IF NOT "
552
+ f"EXISTS` keeps whatever "
553
+ f"is there, so this table came from somewhere else. Refusing here rather than "
554
+ f"at the first insert, which would fail in your request path with an error "
555
+ f"naming a column and not the cause."
556
+ )
557
+ for column, declared in sorted(columns.items()):
558
+ reported = actual[column]
559
+ if reported == declared:
560
+ continue
561
+ raise EngineError(
562
+ f"{table}.{column} is {reported!r} and this map declares it {declared!r}. "
563
+ f"`CREATE TABLE IF NOT EXISTS` keeps a table of that name whatever shape it "
564
+ f"is in, and this library never alters a column's type - so the table came "
565
+ f"from somewhere else, or from a map that rendered this column differently. "
566
+ f"Refusing rather than writing into it: with a timestamp the difference is "
567
+ f"usually precision, and a write that succeeds and comes back rounded is "
568
+ f"worse than one that fails."
569
+ )
570
+ extra = sorted(set(actual) - set(columns))
571
+ if extra:
572
+ log("sde.schema.extra_columns", table=table, columns=extra)
573
+
574
+
575
+ # --- data ------------------------------------------------------------------------------
576
+
577
+ @guarded
578
+ def explain_plan(self, sql: str) -> QueryPlan:
579
+ """Plan an analyst's query without running it, with ``readonly=1`` on every statement.
580
+
581
+ Requirement 19.4. Four questions to the server, all under ``readonly=1`` - which ClickHouse
582
+ refuses a mutation under (code 164) and which leaves the rest working, both halves
583
+ measured, because a write protection that also blocked the measurement would be one nobody
584
+ would keep switched on.
585
+
586
+ The plan tree, ``EXPLAIN ESTIMATE`` for the numbers, ``EXPLAIN QUERY TREE`` for the table
587
+ names, and ``system.tables`` for each table's engine.
588
+
589
+ **The third question exists because of a measurement that broke the obvious design.** The
590
+ table names came from ``EXPLAIN ESTIMATE`` first, and for ``SELECT count() FROM t`` that
591
+ returns **no rows at all** - ClickHouse answers a trivial count from part metadata and
592
+ reads nothing, so there is no table in the estimate. That is precisely the query where
593
+ 9.7's hazard is worst: a count over a ``ReplacingMergeTree`` without ``FINAL`` returns the
594
+ uncollapsed number. So the one query that most needs the warning was the one the estimate
595
+ could not see. ``EXPLAIN QUERY TREE`` names every table in the join tree whatever the read
596
+ optimisation does.
597
+ """
598
+ settings = {"readonly": 1}
599
+ # Read it back. Same reason as the PostgreSQL side: without this, dropping the setting
600
+ # changes nothing observable - nothing EXPLAIN does would write - so the guard would be
601
+ # unverifiable and its removal silent.
602
+ confirmed = self._cx.query("SELECT getSetting('readonly')", settings=settings)
603
+ answer = confirmed.result_rows[0][0] if confirmed.result_rows else None
604
+ if int(answer or 0) < 1:
605
+ raise EngineError(
606
+ f"this server would not accept readonly=1 for planning: getSetting('readonly') "
607
+ f"came back {answer!r}. Refusing to plan anything: requirement 19.2 keeps this "
608
+ f"product out of the data path, and readonly=1 is the whole of what stops a "
609
+ f"statement here from writing - measured, it refuses a mutation with code 164 and "
610
+ f"still answers EXPLAIN ESTIMATE."
611
+ )
612
+ try:
613
+ tree = self._cx.query(f"EXPLAIN {sql}", settings=settings)
614
+ plan = tuple(str(row[0]) for row in tree.result_rows)
615
+ estimate = self._cx.query(f"EXPLAIN ESTIMATE {sql}", settings=settings)
616
+ except Exception as exc:
617
+ raise QueryPlanRefused(
618
+ f"{self.dialect} would not plan this query: {exc} Nothing was executed. This is "
619
+ f"what validating against a live engine means: the control plane checked the query "
620
+ f"against the schema it authored, and this is the schema that exists."
621
+ ) from exc
622
+
623
+ columns = list(estimate.column_names)
624
+ rows = [dict(zip(columns, row, strict=False)) for row in estimate.result_rows]
625
+ values: dict[str, str] = {}
626
+ for name in ("parts", "rows", "marks"):
627
+ if name in columns:
628
+ values[name] = str(sum(int(row.get(name) or 0) for row in rows))
629
+ values["tables"] = str(len(rows))
630
+
631
+ named = self._tables_in(sql, settings=settings) or [
632
+ # The estimate's own names, as a fallback for a server whose analyzer does not answer
633
+ # EXPLAIN QUERY TREE. Both parts come from the answer rather than from this
634
+ # connection's default database: a query can read a table elsewhere, and the default
635
+ # would name the wrong one quietly, because system.tables would simply find nothing.
636
+ (str(row.get("table", "")), str(row.get("database", "")))
637
+ for row in rows
638
+ if row.get("table") and row.get("database")
639
+ ]
640
+ engines: list[tuple[str, str]] = []
641
+ for table, database in named:
642
+ try:
643
+ answer = self._cx.query(
644
+ "SELECT engine FROM system.tables WHERE database = {db:String} "
645
+ "AND name = {tb:String}",
646
+ parameters={"db": database, "tb": table},
647
+ settings=settings,
648
+ )
649
+ except Exception: # pragma: no cover - a client without system.tables access
650
+ # Not fatal and not silent: the plan is still worth having, and the finding this
651
+ # would have produced is a property of the table rather than of the query, so its
652
+ # absence costs a warning rather than a guarantee.
653
+ continue
654
+ for row in answer.result_rows:
655
+ engines.append((table, str(row[0])))
656
+
657
+ return QueryPlan(
658
+ engine=self.dialect,
659
+ dialect=self.dialect,
660
+ plan=plan or ("(the engine returned an empty plan)",),
661
+ cost=Cost(
662
+ units="parts, rows and marks to read",
663
+ basis=(
664
+ "real counts rather than arbitrary units, and coarse below one granule: a "
665
+ "mark covers 8192 rows by default, so a small table reports every row and one "
666
+ "mark whether or not the key filter prunes anything - measured on a "
667
+ "thousand-row table. The figures are what the primary key can rule out before "
668
+ "reading, so they move when the query filters on a key prefix and not when it "
669
+ "filters on anything else. Zero across the board means the engine reads "
670
+ "nothing at all: a trivial count() is answered from part metadata, and the "
671
+ "estimate correctly reports no rows read - which is not the same as a query "
672
+ "that returns nothing."
673
+ ),
674
+ values=values,
675
+ ),
676
+ findings=replacing_merge_tree_finding(engines),
677
+ read_only_enforced=(
678
+ "every statement sent with readonly=1. ClickHouse refuses a mutation under it "
679
+ "(code 164, READONLY) and still answers EXPLAIN ESTIMATE - both measured. There "
680
+ "is no transaction here to roll back, which is the same absence transaction() "
681
+ "refuses rather than pretends about."
682
+ ),
683
+ )
684
+
685
+ def _tables_in(self, sql: str, *, settings: Mapping[str, Any]) -> list[tuple[str, str]]:
686
+ """Every table the query reads, from the analyzer's own tree. ``(table, database)`` pairs.
687
+
688
+ A regex over ``EXPLAIN QUERY TREE`` output, which is the engine's own rendering and could
689
+ change between versions. The cost of that is a **missing** warning rather than a wrong one:
690
+ the finding this feeds is a property of a table, so losing it costs an analyst a sentence
691
+ and costs no guarantee. Worth having anyway, because the alternative loses it for the one
692
+ query where it matters most - see :meth:`explain_plan`.
693
+ """
694
+ import re
695
+
696
+ try:
697
+ tree = self._cx.query(f"EXPLAIN QUERY TREE {sql}", settings=dict(settings))
698
+ except Exception as exc:
699
+ # No pragma here any more, and that is the point. The marker that used to say "no test
700
+ # comes here" sat one line above a log() call whose event name was missing from the
701
+ # vocabulary, and log() raised on an unknown name - so the graceful path was the one
702
+ # that crashed, in the client's process, on an older server.
703
+ log("sde.explain.no_query_tree", reason=str(exc)[:120])
704
+ return []
705
+ found: list[tuple[str, str]] = []
706
+ for row in tree.result_rows:
707
+ for match in re.finditer(r"table_name:\s*(\S+)", str(row[0])):
708
+ qualified = match.group(1)
709
+ database, _, table = qualified.rpartition(".")
710
+ if table and (table, database) not in found:
711
+ found.append((table, database))
712
+ return found
713
+
714
+ @guarded
715
+ def insert(self, table: str, values: Mapping[str, Any]) -> None:
716
+ if not values:
717
+ raise EngineError("nothing to insert")
718
+ cols = sorted(values)
719
+ row = [_as_utc(values[c]) for c in cols]
720
+ try:
721
+ self._cx.insert(_quote(table), [row], column_names=cols)
722
+ except Exception as exc:
723
+ # Surfaced, not swallowed and not rerouted, as in the PostgreSQL adapter. A write that
724
+ # did not happen is not our internal problem to absorb.
725
+ log("sde.write.failed", table=table, error=type(exc).__name__)
726
+ raise EngineError(f"insert into {table} failed: {exc}") from exc
727
+
728
+ @guarded
729
+ def insert_many(self, table: str, rows: Sequence[Mapping[str, Any]]) -> None:
730
+ """One native batch; no promise of cross-block transactional atomicity."""
731
+ cols = batch_columns(rows)
732
+ if not cols:
733
+ return
734
+ data = [[_as_utc(row[c]) for c in cols] for row in rows]
735
+ try:
736
+ self._cx.insert(_quote(table), data, column_names=cols)
737
+ except Exception as exc:
738
+ log("sde.write.failed", table=table, error=type(exc).__name__)
739
+ raise EngineError(f"batch insert into {table} failed: {exc}") from exc
740
+
741
+ @guarded
742
+ def storage_sizes(self, tables: Sequence[str]) -> dict[str, tuple[int, int]]:
743
+ """Each table's bytes and its secondary index bytes, from active parts. Numbers only.
744
+
745
+ One statement: ``system.tables`` says which of the names exist - readable by a runtime
746
+ login without any grant, and only for its own tables - and ``system.parts`` what their
747
+ active parts occupy, which needs the column grant in :data:`STORAGE_COLUMNS`. An empty
748
+ table has no parts and reads 0 bytes; a missing one is absent from the answer, because a
749
+ missing table is not an empty one.
750
+ """
751
+ if not tables:
752
+ return {}
753
+ sql = (
754
+ "SELECT t.name, sum(p.bytes_on_disk), "
755
+ "sum(p.secondary_indices_compressed_bytes + p.secondary_indices_marks_bytes) "
756
+ "FROM system.tables AS t LEFT JOIN ("
757
+ "SELECT table, bytes_on_disk, secondary_indices_compressed_bytes, "
758
+ "secondary_indices_marks_bytes FROM system.parts "
759
+ "WHERE active AND database = currentDatabase()"
760
+ ") AS p ON p.table = t.name "
761
+ "WHERE t.database = currentDatabase() AND t.name IN {tables:Array(String)} "
762
+ "GROUP BY t.name"
763
+ )
764
+ try:
765
+ rows = self._cx.query(
766
+ sql, parameters={"tables": list(tables)}, settings={"join_use_nulls": 0}
767
+ ).result_rows
768
+ except Exception as exc:
769
+ raise EngineError(f"storage sizes could not be read: {exc}") from exc
770
+ return {str(name): (int(total), int(secondary)) for name, total, secondary in rows}
771
+
772
+ @guarded
773
+ def get(self, table: str, key: Mapping[str, Any]) -> dict[str, Any] | None:
774
+ """One row by key, with `FINAL` so a superseded row is never returned.
775
+
776
+ Without `FINAL` a key that has been saved twice returns whichever duplicate the scan
777
+ reaches first until a merge happens - which is to say, nondeterministically the old value.
778
+ Paying for `FINAL` on a point read is the cheaper half of that trade.
779
+ """
780
+ parameters: dict[str, Any] = {}
781
+ where = " AND ".join(
782
+ f"{_quote(column)} = {_query_parameter(f'key_{index}', key[column], parameters)}"
783
+ for index, column in enumerate(sorted(key))
784
+ )
785
+ sql = f"SELECT * FROM {_quote(table)} FINAL WHERE {where} LIMIT 1"
786
+ try:
787
+ result = self._cx.query(sql, parameters=parameters)
788
+ except Exception as exc:
789
+ raise EngineError(f"select from {table} failed: {exc}") from exc
790
+ if not result.result_rows:
791
+ return None
792
+ return _row(result.column_names, result.column_types, result.result_rows[0])
793
+
794
+ # --- rollback protection ------------------------------------------------------------------
795
+ #
796
+ # Append-only and `max()`, which is what makes this identical in both engines: no key to
797
+ # enforce, no row to update, nothing for this engine's lack of a unique constraint to spoil.
798
+
799
+ @guarded
800
+ def map_watermark(self) -> int | None:
801
+ """The highest map version applied against this engine, creating the table if missing.
802
+
803
+ A plain `MergeTree` is right here, and it is the one place in this adapter where that is
804
+ true without qualification: the table is append-only by design and the answer is an
805
+ aggregate, so there are no duplicates to collapse and no reason to pay for `FINAL`.
806
+ """
807
+ try:
808
+ existing = self._cx.query(f"EXISTS TABLE {_quote(WATERMARK_TABLE)}").result_rows
809
+ if not existing or existing[0][0] not in (0, 1):
810
+ raise EngineError("watermark catalog lookup returned no presence result")
811
+ if existing[0][0] == 0:
812
+ self._cx.command(
813
+ f"CREATE TABLE IF NOT EXISTS {_quote(WATERMARK_TABLE)} ("
814
+ f"{_quote('map_version')} Int64, "
815
+ f"{_quote('model_version')} String, "
816
+ f"{_quote('seen_at')} DateTime64(3, 'UTC') DEFAULT now64(3, 'UTC')) "
817
+ f"ENGINE = MergeTree ORDER BY ({_quote('map_version')})"
818
+ )
819
+ result = self._cx.query(
820
+ f"SELECT max({_quote('map_version')}) FROM {_quote(WATERMARK_TABLE)}"
821
+ )
822
+ except Exception as exc:
823
+ raise EngineError(f"reading {WATERMARK_TABLE} failed: {exc}") from exc
824
+ if not result.result_rows or result.result_rows[0][0] is None:
825
+ return None
826
+ highest = int(result.result_rows[0][0])
827
+ # An empty MergeTree answers max() with 0 rather than with null, so zero here means either
828
+ # "no map has been applied" or "map version 0 has been". Map version 0 is what this library
829
+ # reads when a hand-written map omits the field, and a hand-written map is unsigned - so it
830
+ # never reaches this path. Reported as absent, which is the honest reading of the two.
831
+ return highest if highest > 0 else None
832
+
833
+ @guarded
834
+ def record_map_version(self, version: int, *, model_version: str) -> None:
835
+ try:
836
+ self._cx.insert(
837
+ WATERMARK_TABLE,
838
+ [[version, model_version]],
839
+ column_names=["map_version", "model_version"],
840
+ )
841
+ except Exception as exc:
842
+ raise EngineError(
843
+ f"recording a map version in {WATERMARK_TABLE} failed: {exc}"
844
+ ) from exc
845
+
846
+ @guarded
847
+ def select_rows(self, table: str, plan: ReadPlan) -> list[dict[str, Any]]:
848
+ parameters: dict[str, Any] = {}
849
+ def parameter(value: Any) -> str:
850
+ return _query_parameter(f"read_{len(parameters)}", value, parameters)
851
+ statement = read_sql(table, plan, dialect=self.dialect, parameter=parameter)
852
+ try:
853
+ result = self._cx.query(statement, parameters=parameters)
854
+ rows = [_row(result.column_names, result.column_types, row)
855
+ for row in result.result_rows]
856
+ for row in rows:
857
+ for column in plan.columns:
858
+ value = row[column.name]
859
+ if value is not None and column.type in ("timestamp", "timestamptz"):
860
+ instant = (
861
+ value.replace(tzinfo=_dt.UTC)
862
+ if value.tzinfo is None else value.astimezone(_dt.UTC)
863
+ )
864
+ row[column.name] = (
865
+ instant.replace(tzinfo=None) if column.type == "timestamp" else instant
866
+ )
867
+ return [read_row(plan.columns, row) for row in rows]
868
+ except Exception as exc:
869
+ raise EngineError(f"logical scan of {table} failed: {exc}") from exc
870
+
871
+ @guarded
872
+ def count_rows(self, table: str, plan: ReadPlan) -> int:
873
+ parameters: dict[str, Any] = {}
874
+ def parameter(value: Any) -> str:
875
+ return _query_parameter(f"read_{len(parameters)}", value, parameters)
876
+ statement = read_sql(table, plan, dialect=self.dialect, parameter=parameter, count=True)
877
+ try:
878
+ result = self._cx.query(statement, parameters=parameters)
879
+ if not result.result_rows:
880
+ raise EngineError("count query returned no result")
881
+ return int(result.result_rows[0][0])
882
+ except Exception as exc:
883
+ raise EngineError(f"logical count of {table} failed: {exc}") from exc
884
+
885
+ @guarded
886
+ def summarize_rows(
887
+ self, table: str, plan: ReadPlan, column: ReadColumn,
888
+ ) -> Mapping[str, Any]:
889
+ parameters: dict[str, Any] = {}
890
+ def parameter(value: Any) -> str:
891
+ return _query_parameter(f"read_{len(parameters)}", value, parameters)
892
+ statement = summary_sql(table, plan, column, dialect=self.dialect, parameter=parameter)
893
+ try:
894
+ result = self._cx.query(statement, parameters=parameters)
895
+ if not result.result_rows:
896
+ raise EngineError("summary query returned no result")
897
+ return dict(zip(result.column_names, result.result_rows[0], strict=True))
898
+ except Exception as exc:
899
+ raise EngineError(f"logical summary of {table} failed: {exc}") from exc
900
+
901
+ @guarded
902
+ def range(
903
+ self,
904
+ table: str,
905
+ column: str,
906
+ *,
907
+ low: Any = None,
908
+ high: Any = None,
909
+ limit: int | None = None,
910
+ ) -> list[dict[str, Any]]:
911
+ clauses: list[str] = []
912
+ parameters: dict[str, Any] = {}
913
+ if low is not None:
914
+ bound = _query_parameter("low", low, parameters)
915
+ clauses.append(f"{_quote(column)} >= {bound}")
916
+ if high is not None:
917
+ bound = _query_parameter("high", high, parameters)
918
+ clauses.append(f"{_quote(column)} < {bound}")
919
+ where = f" WHERE {' AND '.join(clauses)}" if clauses else ""
920
+ cap = ""
921
+ if limit is not None:
922
+ cap = " LIMIT %(limit)s"
923
+ parameters["limit"] = int(limit)
924
+ sql = f"SELECT * FROM {_quote(table)} FINAL{where} ORDER BY {_quote(column)}{cap}"
925
+ try:
926
+ result = self._cx.query(sql, parameters=parameters)
927
+ except Exception as exc:
928
+ raise EngineError(f"range select from {table} failed: {exc}") from exc
929
+ return [_row(result.column_names, result.column_types, row) for row in result.result_rows]
930
+
931
+ @guarded
932
+ def count(self, table: str) -> int:
933
+ """`FINAL` here too, so this counts entities rather than stored rows.
934
+
935
+ The two differ between a save and the next merge. A count that drifts and then settles is
936
+ worse than a slower count, because it makes a test flaky and a dashboard untrustworthy in
937
+ the same way.
938
+ """
939
+ try:
940
+ result = self._cx.query(f"SELECT count() FROM {_quote(table)} FINAL")
941
+ except Exception as exc:
942
+ raise EngineError(f"count on {table} failed: {exc}") from exc
943
+ return int(result.result_rows[0][0]) if result.result_rows else 0
944
+
945
+ # --- migration ------------------------------------------------------------------------------
946
+ #
947
+ # `sde.migration.Migratable`, the same optional protocol the PostgreSQL adapter satisfies. Two
948
+ # things are different here and neither is smoothed over: reads take `FINAL`, because a key
949
+ # saved twice is an overwrite in this engine rather than an error, and `copy_in` has no
950
+ # conflict clause to write - the collapse *is* the idempotence.
951
+
952
+ @guarded
953
+ def key_range(
954
+ self,
955
+ table: str,
956
+ order: Sequence[str],
957
+ *,
958
+ after: Sequence[Any] | None = None,
959
+ upto: Sequence[Any] | None = None,
960
+ limit: int | None = None,
961
+ ) -> list[dict[str, Any]]:
962
+ """Rows in key order, strictly after one key and up to another inclusive, with `FINAL`.
963
+
964
+ `FINAL` is what makes keyset pagination correct here rather than merely fast enough. Without
965
+ it a key saved twice returns two rows until a merge happens, and the next page starts
966
+ strictly after that key - so one of the duplicates is read and the other is not, which for a
967
+ backfill means copying a row this engine considers superseded.
968
+ """
969
+ cols = key_columns(order, table)
970
+ clauses: list[str] = []
971
+ parameters: dict[str, Any] = {}
972
+ tuple_expr = f"({', '.join(_quote(c) for c in cols)})"
973
+ if after is not None:
974
+ same_width(after, cols, "after")
975
+ bound = [
976
+ _query_parameter(f"after_{index}", value, parameters)
977
+ for index, value in enumerate(after)
978
+ ]
979
+ clauses.append(f"{tuple_expr} > ({', '.join(bound)})")
980
+ if upto is not None:
981
+ same_width(upto, cols, "upto")
982
+ bound = [
983
+ _query_parameter(f"upto_{index}", value, parameters)
984
+ for index, value in enumerate(upto)
985
+ ]
986
+ clauses.append(f"{tuple_expr} <= ({', '.join(bound)})")
987
+ where = f" WHERE {' AND '.join(clauses)}" if clauses else ""
988
+ cap = ""
989
+ if limit is not None:
990
+ cap = " LIMIT %(row_limit)s"
991
+ parameters["row_limit"] = int(limit)
992
+ sql = f"SELECT * FROM {_quote(table)} FINAL{where} ORDER BY {tuple_expr}{cap}"
993
+ try:
994
+ result = self._cx.query(sql, parameters=parameters)
995
+ except Exception as exc:
996
+ raise EngineError(f"key range select from {table} failed: {exc}") from exc
997
+ return [_row(result.column_names, result.column_types, row) for row in result.result_rows]
998
+
999
+ @guarded
1000
+ def nth_key(
1001
+ self, table: str, order: Sequence[str], *, position: int
1002
+ ) -> tuple[Any, ...] | None:
1003
+ """The key of the ``position``-th row in key order, one-based, or None if there is none."""
1004
+ cols = key_columns(order, table)
1005
+ if position < 1:
1006
+ raise EngineError(f"position is one-based; {position} is not a row")
1007
+ projection = ", ".join(_quote(c) for c in cols)
1008
+ sql = (
1009
+ f"SELECT {projection} FROM {_quote(table)} FINAL ORDER BY ({projection}) "
1010
+ f"LIMIT 1 OFFSET %(skip)s"
1011
+ )
1012
+ try:
1013
+ result = self._cx.query(sql, parameters={"skip": position - 1})
1014
+ except Exception as exc:
1015
+ raise EngineError(f"reading row {position} of {table} failed: {exc}") from exc
1016
+ if not result.result_rows:
1017
+ return None
1018
+ row = _row(result.column_names, result.column_types, result.result_rows[0])
1019
+ return tuple(row[column] for column in cols)
1020
+
1021
+ @guarded
1022
+ def copy_in(self, table: str, rows: Sequence[Mapping[str, Any]]) -> None:
1023
+ """Insert rows. Duplicates are collapsed by the table rather than rejected by it.
1024
+
1025
+ There is no `ON CONFLICT` to write and none is needed: the tables this library creates here
1026
+ are `ReplacingMergeTree` ordered by the key, so a row copied twice leaves two parts that
1027
+ collapse to the newest at merge time and read as one under `FINAL`. That is the same
1028
+ idempotence the PostgreSQL path gets from a conflict clause, arrived at from the opposite
1029
+ direction - and it is why a recopied chunk is free in both engines.
1030
+ """
1031
+ if not rows:
1032
+ return
1033
+ cols = sorted(rows[0])
1034
+ for row in rows:
1035
+ if sorted(row) != cols:
1036
+ raise EngineError(
1037
+ f"copy_in into {table} was given rows with different columns "
1038
+ f"({cols} and {sorted(row)}). A chunk comes from one table, so this is a "
1039
+ f"caller assembling it from two."
1040
+ )
1041
+ data = [[_as_utc(row[c]) for c in cols] for row in rows]
1042
+ try:
1043
+ self._cx.insert(_quote(table), data, column_names=cols)
1044
+ except Exception as exc:
1045
+ log("sde.write.failed", table=table, error=type(exc).__name__)
1046
+ raise EngineError(f"copying {len(rows)} rows into {table} failed: {exc}") from exc
1047
+
1048
+ @guarded
1049
+ def backfill_marker(self, *, materialization: str, entity: str) -> int:
1050
+ """How many rows of this entity have been copied into this engine. Zero if none.
1051
+
1052
+ A plain `MergeTree` and `max()`, as the map watermark is - append-only by design, so there
1053
+ are no duplicates to collapse and no reason to pay for `FINAL`. The quirk that needed a
1054
+ comment there is harmless here: an empty aggregate answers 0 rather than null, and 0 is
1055
+ exactly what "nothing has been copied" means, so the two readings coincide.
1056
+ """
1057
+ try:
1058
+ self._cx.command(
1059
+ f"CREATE TABLE IF NOT EXISTS {_quote(BACKFILL_TABLE)} ("
1060
+ f"{_quote('materialization')} String, "
1061
+ f"{_quote('entity')} String, "
1062
+ f"{_quote('rows_copied')} Int64, "
1063
+ f"{_quote('at')} DateTime64(3, 'UTC') DEFAULT now64(3, 'UTC')) "
1064
+ f"ENGINE = MergeTree ORDER BY ({_quote('materialization')}, {_quote('entity')})"
1065
+ )
1066
+ result = self._cx.query(
1067
+ f"SELECT max({_quote('rows_copied')}) FROM {_quote(BACKFILL_TABLE)} "
1068
+ f"WHERE {_quote('materialization')} = %(m)s AND {_quote('entity')} = %(e)s",
1069
+ parameters={"m": materialization, "e": entity},
1070
+ )
1071
+ except Exception as exc:
1072
+ raise EngineError(f"reading {BACKFILL_TABLE} failed: {exc}") from exc
1073
+ if not result.result_rows or result.result_rows[0][0] is None:
1074
+ return 0
1075
+ return int(result.result_rows[0][0])
1076
+
1077
+ @guarded
1078
+ def record_backfill_marker(self, *, materialization: str, entity: str, rows: int) -> None:
1079
+ """Append the new marker. Never update, so an interrupted run leaves a readable trail."""
1080
+ try:
1081
+ self._cx.insert(
1082
+ BACKFILL_TABLE,
1083
+ [[materialization, entity, int(rows)]],
1084
+ column_names=["materialization", "entity", "rows_copied"],
1085
+ )
1086
+ except Exception as exc:
1087
+ raise EngineError(
1088
+ f"recording backfill progress in {BACKFILL_TABLE} failed: {exc}"
1089
+ ) from exc
1090
+
1091
+ # --- transactions ----------------------------------------------------------------------
1092
+
1093
+ def transaction(self) -> Iterator[ClickHouseEngine]:
1094
+ """Refuses. There is no transaction here to give you.
1095
+
1096
+ A no-op context manager would be the friendlier signature and the worse library: the
1097
+ caller would believe a group of writes was atomic, and would find out otherwise from the
1098
+ state of the data rather than from an exception.
1099
+
1100
+ The way out is not a flag. Declare the atomicity - `atomic_with` on the entity - and the
1101
+ planner is then obliged to place those entities in one group, and one group is one engine,
1102
+ so it will not be this one.
1103
+
1104
+ Not decorated with `@contextmanager`, unlike the PostgreSQL one. A decorated generator
1105
+ would need a `yield` after the `raise` to keep the type honest, and that statement is
1106
+ unreachable - `mypy --strict` says so, correctly. A plain method that raises fails at the
1107
+ call, which is one frame earlier and reads better in a traceback.
1108
+ """
1109
+ raise EngineError(
1110
+ "ClickHouse has no multi-statement transactions, so this adapter will not pretend to "
1111
+ "start one. If these writes have to commit together, declare it: `atomic_with` on the "
1112
+ "entities makes them one colocation group, one group is one engine, and the planner is "
1113
+ "then not permitted to put them here. A silent no-op context manager would let the "
1114
+ "writes proceed and let you believe they were atomic."
1115
+ )