smart-data-engine-sdk 0.1.0.dev0__py3-none-any.whl
This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
- sde/__init__.py +226 -0
- sde/canonical.py +141 -0
- sde/capabilities.py +62 -0
- sde/engines/__init__.py +0 -0
- sde/engines/clickhouse.py +689 -0
- sde/engines/orderbook.py +454 -0
- sde/engines/postgres.py +672 -0
- sde/entity.py +170 -0
- sde/errors.py +88 -0
- sde/explain.py +300 -0
- sde/groups.py +97 -0
- sde/hashing.py +242 -0
- sde/infer.py +461 -0
- sde/internal.py +90 -0
- sde/layout.py +660 -0
- sde/logging.py +132 -0
- sde/migration.py +820 -0
- sde/model.py +482 -0
- sde/placement.py +818 -0
- sde/py.typed +0 -0
- sde/routing.py +85 -0
- sde/schema.py +370 -0
- sde/session.py +507 -0
- sde/shapes.py +153 -0
- sde/telemetry.py +736 -0
- sde/testing/__init__.py +14 -0
- sde/testing/loader.py +175 -0
- sde/testing/memory.py +318 -0
- sde/types.py +228 -0
- sde/watermark.py +222 -0
- smart_data_engine_sdk-0.1.0.dev0.dist-info/METADATA +152 -0
- smart_data_engine_sdk-0.1.0.dev0.dist-info/RECORD +35 -0
- smart_data_engine_sdk-0.1.0.dev0.dist-info/WHEEL +4 -0
- smart_data_engine_sdk-0.1.0.dev0.dist-info/licenses/LICENSE +201 -0
- smart_data_engine_sdk-0.1.0.dev0.dist-info/licenses/NOTICE +13 -0
sde/layout.py
ADDED
|
@@ -0,0 +1,660 @@
|
|
|
1
|
+
"""Deriving a default physical layout from a model.
|
|
2
|
+
|
|
3
|
+
There is a boundary question here worth answering explicitly, because getting it wrong would either
|
|
4
|
+
leak the paid part into the open one or make the open one useless on its own.
|
|
5
|
+
|
|
6
|
+
The interesting decisions - which engine a group goes to, which indexes earn their cost, when to
|
|
7
|
+
partition, when a second materialisation pays for itself, when to move - are the planner's, and the
|
|
8
|
+
planner is the part clients pay for. None of that is here.
|
|
9
|
+
|
|
10
|
+
What *is* here is the boring, total function from a model to a schema that stores it: table names,
|
|
11
|
+
column names, column types, a primary key, foreign key columns for relations. That has to be in the
|
|
12
|
+
library, because the library has to work without an account (requirement 12.5). A hand-written
|
|
13
|
+
placement map that says ``"layout": {"auto": true}`` gets this, and everything runs with no key and
|
|
14
|
+
no network. A map from the planner carries an explicit layout instead, and then this is not used.
|
|
15
|
+
|
|
16
|
+
The derivation is deterministic and documented, because it reaches a client's database as DDL and
|
|
17
|
+
they will read it.
|
|
18
|
+
"""
|
|
19
|
+
|
|
20
|
+
from __future__ import annotations
|
|
21
|
+
|
|
22
|
+
import re
|
|
23
|
+
import unicodedata
|
|
24
|
+
from collections.abc import Mapping, Sequence
|
|
25
|
+
from dataclasses import dataclass
|
|
26
|
+
from typing import Any, Final
|
|
27
|
+
|
|
28
|
+
from .errors import DeclarationError
|
|
29
|
+
from .groups import Group
|
|
30
|
+
from .model import LogicalModel
|
|
31
|
+
from .placement import PhysicalLayout
|
|
32
|
+
|
|
33
|
+
__all__ = [
|
|
34
|
+
"CLICKHOUSE_TYPES",
|
|
35
|
+
"FIXED_SCHEMA",
|
|
36
|
+
"ORDERBOOK_KEY",
|
|
37
|
+
"ORDERBOOK_SHAPE",
|
|
38
|
+
"ORDERBOOK_TABLE",
|
|
39
|
+
"ORDERBOOK_TYPES",
|
|
40
|
+
"POSTGRES_TYPES",
|
|
41
|
+
"DerivedLayout",
|
|
42
|
+
"can_store",
|
|
43
|
+
"default_layout",
|
|
44
|
+
"denormalized_layout",
|
|
45
|
+
"fixed_schema_mismatch",
|
|
46
|
+
"group_columns",
|
|
47
|
+
"snake_case",
|
|
48
|
+
"stored_types",
|
|
49
|
+
]
|
|
50
|
+
|
|
51
|
+
|
|
52
|
+
_CAMEL_BOUNDARY: Final = re.compile(r"(?<=[a-z0-9])(?=[A-Z])|(?<=[A-Z])(?=[A-Z][a-z])")
|
|
53
|
+
|
|
54
|
+
POSTGRES_TYPES: Final[Mapping[str, str]] = {
|
|
55
|
+
"bool": "boolean",
|
|
56
|
+
"int32": "integer",
|
|
57
|
+
"int64": "bigint",
|
|
58
|
+
"float32": "real",
|
|
59
|
+
"float64": "double precision",
|
|
60
|
+
"string": "text",
|
|
61
|
+
"bytes": "bytea",
|
|
62
|
+
"uuid": "uuid",
|
|
63
|
+
"date": "date",
|
|
64
|
+
"timestamp": "timestamp",
|
|
65
|
+
"timestamptz": "timestamptz",
|
|
66
|
+
"json": "jsonb",
|
|
67
|
+
}
|
|
68
|
+
|
|
69
|
+
CLICKHOUSE_TYPES: Final[Mapping[str, str]] = {
|
|
70
|
+
"bool": "Bool",
|
|
71
|
+
"int32": "Int32",
|
|
72
|
+
"int64": "Int64",
|
|
73
|
+
"float32": "Float32",
|
|
74
|
+
"float64": "Float64",
|
|
75
|
+
"string": "String",
|
|
76
|
+
# `bytes` and `json` are deliberately absent, and their absence is a *refusal* rather than an
|
|
77
|
+
# oversight. `_column_type` raises for an unmapped neutral type, so a model with either field
|
|
78
|
+
# cannot have a ClickHouse layout derived - which means the planner cannot place that group here
|
|
79
|
+
# and the failure lands at map-build time, in our process, loudly. Both were measured before
|
|
80
|
+
# being given up on:
|
|
81
|
+
#
|
|
82
|
+
# `bytes`: a ClickHouse `String` stores the bytes correctly - `hex()` and `length()` on the
|
|
83
|
+
# server confirm all four bytes of b"\x00\x01\xff\xfe" arrive intact. The read lies. The driver
|
|
84
|
+
# decodes a `String` column to `str`, cannot decode invalid UTF-8, and returns the hex text
|
|
85
|
+
# "0001fffe" instead. Nothing distinguishes a binary `String` column from a text one on the way
|
|
86
|
+
# back, so the adapter cannot correct it, and silently handing a client hex text where they
|
|
87
|
+
# wrote bytes is the kind of corruption that surfaces years later in a checksum.
|
|
88
|
+
#
|
|
89
|
+
# `json`: PostgreSQL `jsonb` returns a parsed `dict`; a ClickHouse `String` returns the
|
|
90
|
+
# original text. The same field would change Python type when its group moved, which breaks the
|
|
91
|
+
# promise the whole product is built on. The native `JSON` type would fix it and was
|
|
92
|
+
# experimental in the server versions in scope, and a signed placement map has to outlive a
|
|
93
|
+
# server upgrade. Resolving this properly means deciding what the neutral `json` type promises
|
|
94
|
+
# on the way *back* - a dict or the exact text - and changing both adapters together. It is a
|
|
95
|
+
# task, not a mapping.
|
|
96
|
+
"uuid": "UUID",
|
|
97
|
+
# Date32 rather than Date: Date covers 1970-2149, which is a range a business date can leave.
|
|
98
|
+
# Silently clamping a date is worse than storing four bytes more.
|
|
99
|
+
"date": "Date32",
|
|
100
|
+
# **Six digits, to match PostgreSQL exactly.** This said three until 7 September 2026, and the
|
|
101
|
+
# comment justifying three compared it against plain `DateTime`, which is second-resolution:
|
|
102
|
+
# true, and the wrong comparison. The engine standing beside it in this product keeps six, so a
|
|
103
|
+
# copy between them changed essentially every row - `datetime.now()` has microseconds - and it
|
|
104
|
+
# changed them silently, because the insert succeeds and the value comes back rounded. The
|
|
105
|
+
# refusal that catches that (`sde.precision_refusal`) then made the product's central shape
|
|
106
|
+
# unexpressible: a transactional group in PostgreSQL with an analytical copy in ClickHouse
|
|
107
|
+
# cannot have a time column, and an order without one is not an order.
|
|
108
|
+
#
|
|
109
|
+
# Measured on ClickHouse 24.8 before the change: `DateTime64(6)` returns
|
|
110
|
+
# `2026-11-09 09:30:15.123456` unchanged where `(3)` returns `.123`, and six digits still cover
|
|
111
|
+
# 1900 to 2261. `DateTime64` is an Int64 tick count whatever the precision, so this is not a
|
|
112
|
+
# storage trade - that part is the type's documented shape rather than something measured here.
|
|
113
|
+
#
|
|
114
|
+
# The bookkeeping tables this library creates for itself keep three, deliberately: they record
|
|
115
|
+
# when we last saw something, they are never copied between engines and never compared against
|
|
116
|
+
# a layout, and changing them would invalidate the ones already on disk for no benefit.
|
|
117
|
+
"timestamp": "DateTime64(6)",
|
|
118
|
+
"timestamptz": "DateTime64(6, 'UTC')",
|
|
119
|
+
|
|
120
|
+
}
|
|
121
|
+
|
|
122
|
+
|
|
123
|
+
def snake_case(name: str) -> str:
|
|
124
|
+
"""``OrderLine`` -> ``order_line``, and NFC-normalised.
|
|
125
|
+
|
|
126
|
+
Normalisation matters for the same reason it matters in the canonical encoding: an identifier
|
|
127
|
+
written with a combining accent and one written composed would otherwise produce two different
|
|
128
|
+
table names for the same entity, and only one of them would have the data in it.
|
|
129
|
+
"""
|
|
130
|
+
normalised = unicodedata.normalize("NFC", name)
|
|
131
|
+
return _CAMEL_BOUNDARY.sub("_", normalised).lower()
|
|
132
|
+
|
|
133
|
+
|
|
134
|
+
ORDERBOOK_TYPES: Final[Mapping[str, str]] = {
|
|
135
|
+
# Only the types the fixed shape uses, and their engine spelling is the C type the API takes.
|
|
136
|
+
# Everything else is absent, and the absence is the same refusal ClickHouse makes for `bytes`:
|
|
137
|
+
# a neutral type with no entry here cannot be given a column, so the group cannot be placed.
|
|
138
|
+
# It matters less here than there, because a model with an extra *field* is already refused by
|
|
139
|
+
# `fixed_schema_mismatch` - but the two refusals have to agree, and the cheapest way to make
|
|
140
|
+
# them agree is for this table to hold exactly the types the shape names.
|
|
141
|
+
"string": "char*",
|
|
142
|
+
"int32": "uint32_t",
|
|
143
|
+
"int64": "int64_t",
|
|
144
|
+
}
|
|
145
|
+
|
|
146
|
+
_DIALECT_TYPES: Final[Mapping[str, Mapping[str, str]]] = {
|
|
147
|
+
"postgres": POSTGRES_TYPES,
|
|
148
|
+
"clickhouse": CLICKHOUSE_TYPES,
|
|
149
|
+
"orderbook": ORDERBOOK_TYPES,
|
|
150
|
+
}
|
|
151
|
+
|
|
152
|
+
# The dialects this library knows, as one public list. It exists because the control plane kept its
|
|
153
|
+
# own vocabulary - `postgresql` there, `postgres` here - and the two agreed only for as long as
|
|
154
|
+
# nothing joined them. The first code that needed both spellings to match was the one rendering DDL
|
|
155
|
+
# for an engine named in the registry, and it failed at runtime rather than at any earlier point.
|
|
156
|
+
DIALECTS: Final[tuple[str, ...]] = ("clickhouse", "orderbook", "postgres")
|
|
157
|
+
|
|
158
|
+
_DECIMAL: Final[Mapping[str, str]] = {
|
|
159
|
+
"postgres": "numeric({digits},{scale})",
|
|
160
|
+
"clickhouse": "Decimal({digits}, {scale})",
|
|
161
|
+
}
|
|
162
|
+
|
|
163
|
+
|
|
164
|
+
# ── An engine whose schema is not ours to choose ─────────────────────────────────────────────────
|
|
165
|
+
#
|
|
166
|
+
# Everything above answers "what columns should this entity get here". The orderbook engine answers
|
|
167
|
+
# it first: it stores L2 depth in one shape, fixed in C++, and there is no CREATE TABLE to send it.
|
|
168
|
+
# So the relationship inverts. For PostgreSQL and ClickHouse the client declares a model and we
|
|
169
|
+
# decide the physical schema; here the engine has already decided, and either the client's model
|
|
170
|
+
# *is* that shape or the group cannot go there.
|
|
171
|
+
#
|
|
172
|
+
# That is stated rather than smoothed over, because smoothing it over has only bad forms. Mapping
|
|
173
|
+
# the client's field names onto the engine's - `at` onto `timestamp_ns`, `qty` onto `quantity` -
|
|
174
|
+
# would mean guessing which declared field is the price from what it is called, and reasoning from a
|
|
175
|
+
# name is the one thing this product refuses everywhere else. Accepting a wider model and dropping
|
|
176
|
+
# the extra fields would lose data in an engine chosen for not losing any.
|
|
177
|
+
#
|
|
178
|
+
# The consequence is a constraint on the model, and the refusal names the whole expected shape so it
|
|
179
|
+
# is actionable in one read.
|
|
180
|
+
|
|
181
|
+
ORDERBOOK_TABLE: Final[str] = "orderbook"
|
|
182
|
+
"""The engine's own name for its storage. One table, and we did not name it."""
|
|
183
|
+
|
|
184
|
+
ORDERBOOK_SHAPE: Final[Mapping[str, str]] = {
|
|
185
|
+
# The address. Not columns in a result row - the engine's query language takes them in the FROM
|
|
186
|
+
# clause, `FROM 'BTCUSDT'.'binance'` - but they are fields of the entity and part of its key,
|
|
187
|
+
# because two rows differing only in symbol are different rows.
|
|
188
|
+
"symbol": "string",
|
|
189
|
+
"exchange": "string",
|
|
190
|
+
# The row.
|
|
191
|
+
"timestamp_ns": "int64",
|
|
192
|
+
"side": "string",
|
|
193
|
+
"level": "int32",
|
|
194
|
+
"price": "int64",
|
|
195
|
+
"quantity": "int64",
|
|
196
|
+
"order_count": "int32",
|
|
197
|
+
# Nullable in the model and *unknown* rather than zero when the engine cannot supply it. The
|
|
198
|
+
# engine returns 0 for a row whose sequence number it does not have, which is a safe sentinel
|
|
199
|
+
# there because its own numbering starts at 1 - and it is not safe here, because a client
|
|
200
|
+
# comparing sequence numbers cannot tell a sentinel from a value. The adapter converts.
|
|
201
|
+
"sequence_number": "int64",
|
|
202
|
+
}
|
|
203
|
+
"""The nine fields an entity must declare, by name and by neutral type, to live in this engine.
|
|
204
|
+
|
|
205
|
+
``price`` and ``quantity`` are integers in the engine's sub-unit, not decimals. That is the engine's
|
|
206
|
+
choice and it is the right one for an orderbook - a decimal per level per update is a rounding
|
|
207
|
+
question in the hot path - but it means a client who declared ``decimal(12,2)`` is declaring a
|
|
208
|
+
different model, and this refuses rather than converting. Converting would put our arithmetic
|
|
209
|
+
between the client's price and their storage.
|
|
210
|
+
"""
|
|
211
|
+
|
|
212
|
+
ORDERBOOK_KEY: Final[tuple[str, ...]] = (
|
|
213
|
+
"symbol",
|
|
214
|
+
"exchange",
|
|
215
|
+
"timestamp_ns",
|
|
216
|
+
"side",
|
|
217
|
+
"level",
|
|
218
|
+
)
|
|
219
|
+
"""The key, in the order the engine addresses by. Positional and load-bearing, as in ClickHouse."""
|
|
220
|
+
|
|
221
|
+
FIXED_SCHEMA: Final[frozenset[str]] = frozenset({"orderbook"})
|
|
222
|
+
"""Dialects whose physical schema the engine imposes rather than accepting from us.
|
|
223
|
+
|
|
224
|
+
A named set rather than a check on the dialect string, so that the three places that have to behave
|
|
225
|
+
differently - the layout, the DDL renderer and the control plane's eligibility filter - agree by
|
|
226
|
+
construction instead of each testing for one name.
|
|
227
|
+
"""
|
|
228
|
+
|
|
229
|
+
_FIXED_SHAPES: Final[Mapping[str, tuple[str, Mapping[str, str], tuple[str, ...]]]] = {
|
|
230
|
+
"orderbook": (ORDERBOOK_TABLE, ORDERBOOK_SHAPE, ORDERBOOK_KEY),
|
|
231
|
+
}
|
|
232
|
+
|
|
233
|
+
assert set(_FIXED_SHAPES) == set(FIXED_SCHEMA), sorted(set(_FIXED_SHAPES) ^ set(FIXED_SCHEMA))
|
|
234
|
+
|
|
235
|
+
|
|
236
|
+
def fixed_schema_mismatch(
|
|
237
|
+
columns: Mapping[str, Mapping[str, str]], *, dialect: str
|
|
238
|
+
) -> str | None:
|
|
239
|
+
"""Why a group cannot live in a fixed-schema engine, or ``None`` if it can.
|
|
240
|
+
|
|
241
|
+
``columns`` is entity name to column name to neutral type - what :func:`group_columns` returns.
|
|
242
|
+
``None`` for a dialect that is not fixed-schema, because "no objection" is the honest answer to
|
|
243
|
+
a question that does not apply: the caller asks this once per engine and branching on the
|
|
244
|
+
dialect at the call site would put the set of fixed-schema engines in two places.
|
|
245
|
+
|
|
246
|
+
A type check alone is not enough and that is the whole reason this exists. The orderbook shape
|
|
247
|
+
is made of ``string``, ``int32`` and ``int64``, all of which every engine can store - so a model
|
|
248
|
+
of two integers and a string passes representability and then fails while the layout is built,
|
|
249
|
+
which is the failure this function was added to stop happening twice.
|
|
250
|
+
"""
|
|
251
|
+
fixed = _FIXED_SHAPES.get(dialect)
|
|
252
|
+
if fixed is None:
|
|
253
|
+
return None
|
|
254
|
+
_, shape, _ = fixed
|
|
255
|
+
|
|
256
|
+
if len(columns) != 1:
|
|
257
|
+
return (
|
|
258
|
+
f"this engine stores one thing, and this group has {len(columns)} entities "
|
|
259
|
+
f"({sorted(columns)}). A colocation group is what shares an engine, so a group of two "
|
|
260
|
+
f"cannot go somewhere with room for one."
|
|
261
|
+
)
|
|
262
|
+
entity, declared = next(iter(columns.items()))
|
|
263
|
+
|
|
264
|
+
missing = sorted(set(shape) - set(declared))
|
|
265
|
+
extra = sorted(set(declared) - set(shape))
|
|
266
|
+
wrong = sorted(
|
|
267
|
+
f"{name} is declared {declared[name]!r} and this engine stores {shape[name]!r}"
|
|
268
|
+
for name in sorted(set(shape) & set(declared))
|
|
269
|
+
if declared[name] != shape[name]
|
|
270
|
+
)
|
|
271
|
+
if not (missing or extra or wrong):
|
|
272
|
+
return None
|
|
273
|
+
|
|
274
|
+
problems = []
|
|
275
|
+
if missing:
|
|
276
|
+
problems.append(f"{entity} declares no {missing}")
|
|
277
|
+
if extra:
|
|
278
|
+
problems.append(f"{entity} declares {extra}, which this engine has nowhere to put")
|
|
279
|
+
if wrong:
|
|
280
|
+
problems.append("; ".join(wrong))
|
|
281
|
+
expected = ", ".join(f"{name}: {kind}" for name, kind in shape.items())
|
|
282
|
+
return (
|
|
283
|
+
f"{'. '.join(problems)}. This engine's schema is fixed in the engine and not chosen by us, "
|
|
284
|
+
f"so a model either is that shape or cannot be stored here. The shape is exactly: "
|
|
285
|
+
f"{expected}."
|
|
286
|
+
)
|
|
287
|
+
|
|
288
|
+
|
|
289
|
+
def _column_type(neutral: str, dialect: str) -> str:
|
|
290
|
+
"""One neutral type, one engine type, and no defaulting.
|
|
291
|
+
|
|
292
|
+
A missing entry raises rather than falling back to a string, because a type nobody mapped is a
|
|
293
|
+
gap in an adapter and the honest place to find that out is here - not in a client's database,
|
|
294
|
+
where the column already exists with the wrong type and changing it is a migration.
|
|
295
|
+
"""
|
|
296
|
+
types = _DIALECT_TYPES[dialect]
|
|
297
|
+
if neutral.startswith("decimal("):
|
|
298
|
+
digits, scale = neutral[len("decimal(") : -1].split(",")
|
|
299
|
+
template = _DECIMAL.get(dialect)
|
|
300
|
+
if template is None:
|
|
301
|
+
# A DeclarationError rather than a KeyError, because this is the same answer as any
|
|
302
|
+
# other unmapped neutral type - the engine cannot give this column a type - and
|
|
303
|
+
# `can_store` distinguishes "cannot store" from "does not parse" by which exception it
|
|
304
|
+
# sees. The orderbook engine stores prices as integers in a sub-unit; a decimal there
|
|
305
|
+
# would need our arithmetic between the client's price and their storage.
|
|
306
|
+
raise DeclarationError(
|
|
307
|
+
f"no {dialect} type for {neutral!r}. This engine has no decimal type at all, so "
|
|
308
|
+
f"this is a gap in the model for this engine rather than in the adapter."
|
|
309
|
+
)
|
|
310
|
+
return template.format(digits=digits, scale=scale)
|
|
311
|
+
try:
|
|
312
|
+
return types[neutral]
|
|
313
|
+
except KeyError:
|
|
314
|
+
raise DeclarationError(
|
|
315
|
+
f"no {dialect} type for {neutral!r}. Every member of the neutral vocabulary needs one; "
|
|
316
|
+
"this is a gap in the adapter rather than a problem with the model."
|
|
317
|
+
) from None
|
|
318
|
+
|
|
319
|
+
|
|
320
|
+
def can_store(neutral: str, *, dialect: str) -> bool:
|
|
321
|
+
"""Whether this dialect has a column type for this neutral type.
|
|
322
|
+
|
|
323
|
+
The one public question about representability, and it exists because the answer was previously
|
|
324
|
+
only obtainable by trying: a group with a ``bytes`` field placed in ClickHouse produced a
|
|
325
|
+
valid scoring decision and then raised out of ``default_layout``. The refusal was correct and it
|
|
326
|
+
landed in the wrong place - at map-build time, where it reads as our defect rather than as a
|
|
327
|
+
reason one engine was not a candidate.
|
|
328
|
+
|
|
329
|
+
Implemented by asking the renderer rather than by a second table. A representability check that
|
|
330
|
+
could disagree with the type mapping would be worse than none: it would let a placement be
|
|
331
|
+
approved and then fail to apply, which is the failure the byte contract exists to prevent.
|
|
332
|
+
|
|
333
|
+
A malformed type - ``decimal(x)`` - is not answered ``False``. "This engine cannot store it"
|
|
334
|
+
is a claim about the engine; a type that parses nowhere is a claim about the model, and the
|
|
335
|
+
model's own validation owns it. So that raises through.
|
|
336
|
+
"""
|
|
337
|
+
if dialect not in _DIALECT_TYPES:
|
|
338
|
+
raise DeclarationError(
|
|
339
|
+
f"no type table for dialect {dialect!r}; this library knows {sorted(_DIALECT_TYPES)}. "
|
|
340
|
+
f"Answering False would say 'that engine cannot store it' about an engine this library "
|
|
341
|
+
f"has never heard of."
|
|
342
|
+
)
|
|
343
|
+
try:
|
|
344
|
+
_column_type(neutral, dialect)
|
|
345
|
+
except DeclarationError:
|
|
346
|
+
return False
|
|
347
|
+
return True
|
|
348
|
+
|
|
349
|
+
|
|
350
|
+
def _neutral_columns(model: LogicalModel, group: Group) -> dict[str, dict[str, str]]:
|
|
351
|
+
"""Every column a group's tables need, in the neutral vocabulary, before any dialect.
|
|
352
|
+
|
|
353
|
+
Factored out of ``default_layout`` because two things need it and a second copy of the loop is
|
|
354
|
+
how they stop agreeing. ``default_layout`` maps each of these through ``_column_type``;
|
|
355
|
+
``stored_types`` asks which of them an engine can represent at all.
|
|
356
|
+
|
|
357
|
+
A group has a foreign-key column for every relation whose source is a member. Today those add no
|
|
358
|
+
*type* the group did not already have, because a relation unions its two ends into one
|
|
359
|
+
colocation group, so the target's key fields are declared fields of a member. That is a fact
|
|
360
|
+
about how groups are formed rather than about layouts, and this function does not rely on it -
|
|
361
|
+
which is the point of deriving the set here instead of from ``spec.fields`` at the call site.
|
|
362
|
+
|
|
363
|
+
Insertion order is the declared field order, then relations in declared order. Callers that need
|
|
364
|
+
a stable ordering get one without sorting; callers that sort - the DDL renderer does - are
|
|
365
|
+
unaffected.
|
|
366
|
+
"""
|
|
367
|
+
out: dict[str, dict[str, str]] = {}
|
|
368
|
+
for entity_name in group.members:
|
|
369
|
+
spec = model.entity(entity_name)
|
|
370
|
+
cols: dict[str, str] = {field.name: field.type for field in spec.fields}
|
|
371
|
+
for relation in model.relations:
|
|
372
|
+
if relation.source != entity_name:
|
|
373
|
+
continue
|
|
374
|
+
target = model.entity(relation.target)
|
|
375
|
+
for key_field in target.key:
|
|
376
|
+
key_spec = next(f for f in target.fields if f.name == key_field)
|
|
377
|
+
# <relation>_<key field>, for a single-field key and for a composite one alike. An
|
|
378
|
+
# earlier version branched on the key's arity and produced the same name in both
|
|
379
|
+
# arms, which ruff spotted as a useless condition - correctly, and it was a leftover
|
|
380
|
+
# from an idea about naming that turned out not to be worth the inconsistency.
|
|
381
|
+
cols[f"{relation.name}_{key_field}"] = key_spec.type
|
|
382
|
+
out[entity_name] = cols
|
|
383
|
+
return out
|
|
384
|
+
|
|
385
|
+
|
|
386
|
+
def group_columns(model: LogicalModel, group: Group) -> Mapping[str, Mapping[str, str]]:
|
|
387
|
+
"""Every column a group's tables need, per entity, in the neutral vocabulary.
|
|
388
|
+
|
|
389
|
+
The public form of the derivation the layout uses. Exposed because the control plane has two
|
|
390
|
+
questions to ask before it picks an engine - can this engine store these types, and does this
|
|
391
|
+
group *fit* an engine whose shape is fixed - and the second one needs the column names, not just
|
|
392
|
+
the types.
|
|
393
|
+
"""
|
|
394
|
+
return _neutral_columns(model, group)
|
|
395
|
+
|
|
396
|
+
|
|
397
|
+
def stored_types(model: LogicalModel, group: Group) -> tuple[str, ...]:
|
|
398
|
+
"""The neutral types a group's tables need, sorted and deduplicated.
|
|
399
|
+
|
|
400
|
+
For the control plane, which has to answer "can this engine hold this group" *before* it picks
|
|
401
|
+
one. Without this the only available answer was to place the group and watch ``default_layout``
|
|
402
|
+
raise - a correct refusal in the wrong place, because it arrives after the decision is made and
|
|
403
|
+
reads as our defect rather than as the reason an engine was not a candidate.
|
|
404
|
+
|
|
405
|
+
Paired with :func:`can_store`, which is asked once per type. Both come from the same column
|
|
406
|
+
derivation the layout uses, so an engine reported able to hold a group is one whose layout can
|
|
407
|
+
actually be rendered - and that equivalence is what the control plane relies on when it excludes
|
|
408
|
+
an engine instead of discovering the problem while building the map.
|
|
409
|
+
"""
|
|
410
|
+
return tuple(
|
|
411
|
+
sorted(
|
|
412
|
+
{
|
|
413
|
+
column_type
|
|
414
|
+
for cols in _neutral_columns(model, group).values()
|
|
415
|
+
for column_type in cols.values()
|
|
416
|
+
}
|
|
417
|
+
)
|
|
418
|
+
)
|
|
419
|
+
|
|
420
|
+
|
|
421
|
+
def default_layout(
|
|
422
|
+
model: LogicalModel, group: Group, *, dialect: str = "postgres"
|
|
423
|
+
) -> PhysicalLayout:
|
|
424
|
+
"""The obvious schema for a group in one engine.
|
|
425
|
+
|
|
426
|
+
Foreign key columns are named ``<relation>_<target key field>``, which is the convention nearly
|
|
427
|
+
every ORM uses, so a client looking at their own database recognises what they are seeing. A
|
|
428
|
+
relation to an entity with a composite key produces one column per key field.
|
|
429
|
+
"""
|
|
430
|
+
if dialect not in _DIALECT_TYPES:
|
|
431
|
+
raise DeclarationError(
|
|
432
|
+
f"no default layout for dialect {dialect!r}. Adding one is an adapter's job, and it "
|
|
433
|
+
f"has "
|
|
434
|
+
f"to be added deliberately rather than approximated from an existing one - the known "
|
|
435
|
+
f"dialects are {sorted(_DIALECT_TYPES)}."
|
|
436
|
+
)
|
|
437
|
+
|
|
438
|
+
neutral = _neutral_columns(model, group)
|
|
439
|
+
|
|
440
|
+
if dialect in FIXED_SCHEMA:
|
|
441
|
+
mismatch = fixed_schema_mismatch(neutral, dialect=dialect)
|
|
442
|
+
if mismatch is not None:
|
|
443
|
+
raise DeclarationError(mismatch)
|
|
444
|
+
table, shape, _ = _FIXED_SHAPES[dialect]
|
|
445
|
+
entity = next(iter(neutral))
|
|
446
|
+
# The engine's own table name and the engine's own column names, typed through the same
|
|
447
|
+
# `_column_type` as everything else so that an unmapped neutral type still raises here
|
|
448
|
+
# rather than producing a column with a plausible C type nobody chose.
|
|
449
|
+
return PhysicalLayout(
|
|
450
|
+
tables={entity: table},
|
|
451
|
+
columns={entity: {name: _column_type(kind, dialect) for name, kind in shape.items()}},
|
|
452
|
+
indexes=(),
|
|
453
|
+
)
|
|
454
|
+
|
|
455
|
+
tables: dict[str, str] = {}
|
|
456
|
+
columns: dict[str, dict[str, str]] = {}
|
|
457
|
+
indexes: list[dict[str, object]] = []
|
|
458
|
+
|
|
459
|
+
for entity_name in group.members:
|
|
460
|
+
tables[entity_name] = snake_case(entity_name)
|
|
461
|
+
columns[entity_name] = {
|
|
462
|
+
column: _column_type(neutral_type, dialect)
|
|
463
|
+
for column, neutral_type in neutral[entity_name].items()
|
|
464
|
+
}
|
|
465
|
+
|
|
466
|
+
# One index, and only because a foreign key without one turns every relation walk into a
|
|
467
|
+
# sequential scan. Anything beyond this is a planner decision with a cost attached, and the
|
|
468
|
+
# library has no business guessing at it.
|
|
469
|
+
#
|
|
470
|
+
# None at all for ClickHouse, and that is not an omission. It has no B-tree: `CREATE INDEX`
|
|
471
|
+
# there builds a *data-skipping* index, which needs a type and a granularity and helps only
|
|
472
|
+
# when the data is already ordered so whole granules can be ruled out. Choosing those is a
|
|
473
|
+
# planner decision with a cost attached, exactly like the ones above that this function
|
|
474
|
+
# refuses to make. What a MergeTree does have is its `ORDER BY`, which is the primary index
|
|
475
|
+
# and is derived from the declared key by the adapter - so the useful index exists, it is
|
|
476
|
+
# simply not expressed as a row in this list.
|
|
477
|
+
if dialect != "clickhouse":
|
|
478
|
+
for relation in model.relations:
|
|
479
|
+
if relation.source != entity_name:
|
|
480
|
+
continue
|
|
481
|
+
target = model.entity(relation.target)
|
|
482
|
+
indexes.append(
|
|
483
|
+
{
|
|
484
|
+
"entity": entity_name,
|
|
485
|
+
"name": f"{snake_case(entity_name)}_{relation.name}_idx",
|
|
486
|
+
"columns": [f"{relation.name}_{k}" for k in target.key],
|
|
487
|
+
}
|
|
488
|
+
)
|
|
489
|
+
|
|
490
|
+
return PhysicalLayout(tables=tables, columns=columns, indexes=tuple(indexes))
|
|
491
|
+
|
|
492
|
+
|
|
493
|
+
# ── The other layout: one wide table for reading, and the field it will not copy ─────────────────
|
|
494
|
+
|
|
495
|
+
|
|
496
|
+
@dataclass(frozen=True)
|
|
497
|
+
class DerivedLayout:
|
|
498
|
+
"""A denormalised layout, plus what it left behind and why.
|
|
499
|
+
|
|
500
|
+
One value rather than a layout and a second function answering "what was excluded", because two
|
|
501
|
+
derivations of the same decision drift - and the thing that would drift here is a list of
|
|
502
|
+
personal data fields, where a stale answer is the worst kind. The control plane puts ``layout``
|
|
503
|
+
in the map and ``excluded_pii`` in the report; both come from one pass.
|
|
504
|
+
"""
|
|
505
|
+
|
|
506
|
+
layout: PhysicalLayout
|
|
507
|
+
root: str
|
|
508
|
+
"""The entity whose grain the wide table has: one row per root row."""
|
|
509
|
+
|
|
510
|
+
inlined: tuple[tuple[str, str], ...]
|
|
511
|
+
"""(relation, target entity) pairs that were flattened in, in declared order."""
|
|
512
|
+
|
|
513
|
+
excluded_pii: tuple[str, ...]
|
|
514
|
+
"""``Entity.field`` for every personal-data field kept out, sorted.
|
|
515
|
+
|
|
516
|
+
Reported rather than silently absent. An analyst who queries the wide table and finds no email
|
|
517
|
+
needs to be told that the column is missing *by decision*, not conclude the data is incomplete -
|
|
518
|
+
and the client needs to see which fields they would have to allow explicitly.
|
|
519
|
+
"""
|
|
520
|
+
|
|
521
|
+
|
|
522
|
+
def _relation_columns(model: LogicalModel, group: Group) -> dict[str, list[Any]]:
|
|
523
|
+
"""Intra-group relations by source entity, in declared order."""
|
|
524
|
+
out: dict[str, list[Any]] = {name: [] for name in group.members}
|
|
525
|
+
members = set(group.members)
|
|
526
|
+
for relation in model.relations:
|
|
527
|
+
if relation.source in members and relation.target in members:
|
|
528
|
+
out[relation.source].append(relation)
|
|
529
|
+
return out
|
|
530
|
+
|
|
531
|
+
|
|
532
|
+
def denormalized_layout(
|
|
533
|
+
model: LogicalModel,
|
|
534
|
+
group: Group,
|
|
535
|
+
*,
|
|
536
|
+
dialect: str = "clickhouse",
|
|
537
|
+
include_pii: Sequence[str] = (),
|
|
538
|
+
) -> DerivedLayout:
|
|
539
|
+
"""One wide table for a group, for an engine that is read from rather than written to.
|
|
540
|
+
|
|
541
|
+
Requirement 5.1: the same entity may be materialised differently in different engines -
|
|
542
|
+
normalised in a transactional store to write, flattened in an analytical one to read - from one
|
|
543
|
+
logical model. This is the second derivation, and it is here rather than in the control plane
|
|
544
|
+
for the same reason ``default_layout`` is: the library has to be able to apply a hand-written
|
|
545
|
+
map, and a schema derived in two places is a schema that eventually differs in one.
|
|
546
|
+
|
|
547
|
+
**Personal data is excluded unless named in ``include_pii``.** Requirement 5.5, and it lives in
|
|
548
|
+
this function rather than in the planner on purpose: an analytical materialisation is a second
|
|
549
|
+
copy of the data in a different engine with different access controls, so "we do not copy your
|
|
550
|
+
personal data there" is a promise worth being able to *read* rather than take. Fails closed - an
|
|
551
|
+
empty allowance excludes every declared personal-data field, and a caller who wants one copied
|
|
552
|
+
names it. Names are ``Entity.field``, because a group holds several entities and ``email`` alone
|
|
553
|
+
would be ambiguous the moment two of them have one.
|
|
554
|
+
|
|
555
|
+
The rule applies to the root entity's own fields as well as to the inlined ones. "Denormalised
|
|
556
|
+
into an analytical materialisation" is about the copy arriving there, not about which side of a
|
|
557
|
+
join it came from, and a rule with an exception for the root would be a rule nobody could state
|
|
558
|
+
in one sentence.
|
|
559
|
+
|
|
560
|
+
Four refusals, each because the friendly alternative produces something worse than an error:
|
|
561
|
+
|
|
562
|
+
**A group with no intra-group relation.** There is nothing to flatten, so the wide table would
|
|
563
|
+
be ``default_layout`` under another name: a second copy costing storage, replication and lag
|
|
564
|
+
while answering exactly the questions the source already answers.
|
|
565
|
+
|
|
566
|
+
**An ambiguous root.** The grain is the entity nothing else in the group points at. If two
|
|
567
|
+
entities qualify, one row of the wide table would mean whatever this function guessed, and a
|
|
568
|
+
table whose row nobody chose the meaning of is worse than no table.
|
|
569
|
+
|
|
570
|
+
**A fixed-schema engine.** Its schema is not ours to choose, so there is no wide table to make
|
|
571
|
+
there - see ``FIXED_SCHEMA``.
|
|
572
|
+
|
|
573
|
+
**An allowance naming something that is not a declared personal-data field of this group.** Both
|
|
574
|
+
directions: a typo would leave the field excluded while the client believed they had allowed it,
|
|
575
|
+
and a name that is not personal data at all suggests the client thinks it is.
|
|
576
|
+
"""
|
|
577
|
+
if dialect in FIXED_SCHEMA:
|
|
578
|
+
raise DeclarationError(
|
|
579
|
+
f"{dialect!r} imposes its own schema, so there is no denormalised layout to derive for "
|
|
580
|
+
f"it. A wide table is a choice about physical shape, and this engine has already made "
|
|
581
|
+
f"it."
|
|
582
|
+
)
|
|
583
|
+
|
|
584
|
+
pii_by_entity = {
|
|
585
|
+
name: set(model.entity(name).pii) for name in group.members
|
|
586
|
+
}
|
|
587
|
+
known_pii = sorted(
|
|
588
|
+
f"{entity}.{field}" for entity, fields in pii_by_entity.items() for field in fields
|
|
589
|
+
)
|
|
590
|
+
allowed = tuple(include_pii)
|
|
591
|
+
unknown = sorted(set(allowed) - set(known_pii))
|
|
592
|
+
if unknown:
|
|
593
|
+
raise DeclarationError(
|
|
594
|
+
f"include_pii names {unknown}, which {group.name} does not declare as personal data. "
|
|
595
|
+
f"Refused in both directions rather than ignored: a misspelling would leave the field "
|
|
596
|
+
f"excluded while you believed you had allowed it, and a name that is not personal data "
|
|
597
|
+
f"suggests it was expected to be. This group declares {known_pii or 'none'}."
|
|
598
|
+
)
|
|
599
|
+
|
|
600
|
+
relations = _relation_columns(model, group)
|
|
601
|
+
targets = {r.target for outgoing in relations.values() for r in outgoing}
|
|
602
|
+
roots = sorted(name for name in group.members if name not in targets)
|
|
603
|
+
if not any(relations.values()):
|
|
604
|
+
raise DeclarationError(
|
|
605
|
+
f"{group.name} has no relation inside it, so there is nothing to denormalise. A wide "
|
|
606
|
+
f"table here would be the default layout under another name - a second copy paying for "
|
|
607
|
+
f"storage, replication and lag to answer the questions the source already answers."
|
|
608
|
+
)
|
|
609
|
+
if len(roots) != 1:
|
|
610
|
+
raise DeclarationError(
|
|
611
|
+
f"{group.name} has {roots} as candidate roots and a wide table has one grain. Refusing "
|
|
612
|
+
f"rather than picking: one row would mean whatever this function chose, and a table "
|
|
613
|
+
f"whose row nobody decided the meaning of is worse than no table."
|
|
614
|
+
)
|
|
615
|
+
root = roots[0]
|
|
616
|
+
|
|
617
|
+
neutral = _neutral_columns(model, group)
|
|
618
|
+
columns: dict[str, str] = {}
|
|
619
|
+
excluded: list[str] = []
|
|
620
|
+
|
|
621
|
+
def take(entity: str, field: str, column: str, neutral_type: str) -> None:
|
|
622
|
+
qualified = f"{entity}.{field}"
|
|
623
|
+
if field in pii_by_entity[entity] and qualified not in allowed:
|
|
624
|
+
excluded.append(qualified)
|
|
625
|
+
return
|
|
626
|
+
columns[column] = _column_type(neutral_type, dialect)
|
|
627
|
+
|
|
628
|
+
# The root's own columns keep their names. A column belonging to an inlined relation is skipped
|
|
629
|
+
# here and taken on the inlined pass instead, and that is **not** about duplicates: the foreign
|
|
630
|
+
# key `user_id` and the inlined `User.id` produce the same column name by construction, so the
|
|
631
|
+
# dictionary assignment would be idempotent either way. It is about the personal-data rule. A
|
|
632
|
+
# foreign-key column is named after the relation, not after a field of the root, so the check in
|
|
633
|
+
# `take()` does not recognise it - and a target whose *key* is declared personal data
|
|
634
|
+
# (`Person.national_id`) would arrive here under `person_national_id` with nothing to stop it.
|
|
635
|
+
# Deferring to the inlined pass puts the column under the entity that declared it.
|
|
636
|
+
inlined_names = {relation.name for relation in relations[root]}
|
|
637
|
+
for column, neutral_type in neutral[root].items():
|
|
638
|
+
if any(column.startswith(f"{name}_") for name in inlined_names):
|
|
639
|
+
continue
|
|
640
|
+
take(root, column, column, neutral_type)
|
|
641
|
+
|
|
642
|
+
inlined: list[tuple[str, str]] = []
|
|
643
|
+
for relation in relations[root]:
|
|
644
|
+
target = relation.target
|
|
645
|
+
inlined.append((relation.name, target))
|
|
646
|
+
for column, neutral_type in neutral[target].items():
|
|
647
|
+
# `<relation>_<field>`, the same convention the foreign key already uses, so a client
|
|
648
|
+
# reading their own analytical table recognises where a column came from without a map.
|
|
649
|
+
take(target, column, f"{relation.name}_{column}", neutral_type)
|
|
650
|
+
|
|
651
|
+
return DerivedLayout(
|
|
652
|
+
layout=PhysicalLayout(
|
|
653
|
+
tables={root: f"{snake_case(root)}_wide"},
|
|
654
|
+
columns={root: columns},
|
|
655
|
+
indexes=(),
|
|
656
|
+
),
|
|
657
|
+
root=root,
|
|
658
|
+
inlined=tuple(inlined),
|
|
659
|
+
excluded_pii=tuple(sorted(excluded)),
|
|
660
|
+
)
|