smart-data-engine-sdk 0.1.0.dev0__py3-none-any.whl

This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
sde/layout.py ADDED
@@ -0,0 +1,660 @@
1
+ """Deriving a default physical layout from a model.
2
+
3
+ There is a boundary question here worth answering explicitly, because getting it wrong would either
4
+ leak the paid part into the open one or make the open one useless on its own.
5
+
6
+ The interesting decisions - which engine a group goes to, which indexes earn their cost, when to
7
+ partition, when a second materialisation pays for itself, when to move - are the planner's, and the
8
+ planner is the part clients pay for. None of that is here.
9
+
10
+ What *is* here is the boring, total function from a model to a schema that stores it: table names,
11
+ column names, column types, a primary key, foreign key columns for relations. That has to be in the
12
+ library, because the library has to work without an account (requirement 12.5). A hand-written
13
+ placement map that says ``"layout": {"auto": true}`` gets this, and everything runs with no key and
14
+ no network. A map from the planner carries an explicit layout instead, and then this is not used.
15
+
16
+ The derivation is deterministic and documented, because it reaches a client's database as DDL and
17
+ they will read it.
18
+ """
19
+
20
+ from __future__ import annotations
21
+
22
+ import re
23
+ import unicodedata
24
+ from collections.abc import Mapping, Sequence
25
+ from dataclasses import dataclass
26
+ from typing import Any, Final
27
+
28
+ from .errors import DeclarationError
29
+ from .groups import Group
30
+ from .model import LogicalModel
31
+ from .placement import PhysicalLayout
32
+
33
+ __all__ = [
34
+ "CLICKHOUSE_TYPES",
35
+ "FIXED_SCHEMA",
36
+ "ORDERBOOK_KEY",
37
+ "ORDERBOOK_SHAPE",
38
+ "ORDERBOOK_TABLE",
39
+ "ORDERBOOK_TYPES",
40
+ "POSTGRES_TYPES",
41
+ "DerivedLayout",
42
+ "can_store",
43
+ "default_layout",
44
+ "denormalized_layout",
45
+ "fixed_schema_mismatch",
46
+ "group_columns",
47
+ "snake_case",
48
+ "stored_types",
49
+ ]
50
+
51
+
52
+ _CAMEL_BOUNDARY: Final = re.compile(r"(?<=[a-z0-9])(?=[A-Z])|(?<=[A-Z])(?=[A-Z][a-z])")
53
+
54
+ POSTGRES_TYPES: Final[Mapping[str, str]] = {
55
+ "bool": "boolean",
56
+ "int32": "integer",
57
+ "int64": "bigint",
58
+ "float32": "real",
59
+ "float64": "double precision",
60
+ "string": "text",
61
+ "bytes": "bytea",
62
+ "uuid": "uuid",
63
+ "date": "date",
64
+ "timestamp": "timestamp",
65
+ "timestamptz": "timestamptz",
66
+ "json": "jsonb",
67
+ }
68
+
69
+ CLICKHOUSE_TYPES: Final[Mapping[str, str]] = {
70
+ "bool": "Bool",
71
+ "int32": "Int32",
72
+ "int64": "Int64",
73
+ "float32": "Float32",
74
+ "float64": "Float64",
75
+ "string": "String",
76
+ # `bytes` and `json` are deliberately absent, and their absence is a *refusal* rather than an
77
+ # oversight. `_column_type` raises for an unmapped neutral type, so a model with either field
78
+ # cannot have a ClickHouse layout derived - which means the planner cannot place that group here
79
+ # and the failure lands at map-build time, in our process, loudly. Both were measured before
80
+ # being given up on:
81
+ #
82
+ # `bytes`: a ClickHouse `String` stores the bytes correctly - `hex()` and `length()` on the
83
+ # server confirm all four bytes of b"\x00\x01\xff\xfe" arrive intact. The read lies. The driver
84
+ # decodes a `String` column to `str`, cannot decode invalid UTF-8, and returns the hex text
85
+ # "0001fffe" instead. Nothing distinguishes a binary `String` column from a text one on the way
86
+ # back, so the adapter cannot correct it, and silently handing a client hex text where they
87
+ # wrote bytes is the kind of corruption that surfaces years later in a checksum.
88
+ #
89
+ # `json`: PostgreSQL `jsonb` returns a parsed `dict`; a ClickHouse `String` returns the
90
+ # original text. The same field would change Python type when its group moved, which breaks the
91
+ # promise the whole product is built on. The native `JSON` type would fix it and was
92
+ # experimental in the server versions in scope, and a signed placement map has to outlive a
93
+ # server upgrade. Resolving this properly means deciding what the neutral `json` type promises
94
+ # on the way *back* - a dict or the exact text - and changing both adapters together. It is a
95
+ # task, not a mapping.
96
+ "uuid": "UUID",
97
+ # Date32 rather than Date: Date covers 1970-2149, which is a range a business date can leave.
98
+ # Silently clamping a date is worse than storing four bytes more.
99
+ "date": "Date32",
100
+ # **Six digits, to match PostgreSQL exactly.** This said three until 7 September 2026, and the
101
+ # comment justifying three compared it against plain `DateTime`, which is second-resolution:
102
+ # true, and the wrong comparison. The engine standing beside it in this product keeps six, so a
103
+ # copy between them changed essentially every row - `datetime.now()` has microseconds - and it
104
+ # changed them silently, because the insert succeeds and the value comes back rounded. The
105
+ # refusal that catches that (`sde.precision_refusal`) then made the product's central shape
106
+ # unexpressible: a transactional group in PostgreSQL with an analytical copy in ClickHouse
107
+ # cannot have a time column, and an order without one is not an order.
108
+ #
109
+ # Measured on ClickHouse 24.8 before the change: `DateTime64(6)` returns
110
+ # `2026-11-09 09:30:15.123456` unchanged where `(3)` returns `.123`, and six digits still cover
111
+ # 1900 to 2261. `DateTime64` is an Int64 tick count whatever the precision, so this is not a
112
+ # storage trade - that part is the type's documented shape rather than something measured here.
113
+ #
114
+ # The bookkeeping tables this library creates for itself keep three, deliberately: they record
115
+ # when we last saw something, they are never copied between engines and never compared against
116
+ # a layout, and changing them would invalidate the ones already on disk for no benefit.
117
+ "timestamp": "DateTime64(6)",
118
+ "timestamptz": "DateTime64(6, 'UTC')",
119
+
120
+ }
121
+
122
+
123
+ def snake_case(name: str) -> str:
124
+ """``OrderLine`` -> ``order_line``, and NFC-normalised.
125
+
126
+ Normalisation matters for the same reason it matters in the canonical encoding: an identifier
127
+ written with a combining accent and one written composed would otherwise produce two different
128
+ table names for the same entity, and only one of them would have the data in it.
129
+ """
130
+ normalised = unicodedata.normalize("NFC", name)
131
+ return _CAMEL_BOUNDARY.sub("_", normalised).lower()
132
+
133
+
134
+ ORDERBOOK_TYPES: Final[Mapping[str, str]] = {
135
+ # Only the types the fixed shape uses, and their engine spelling is the C type the API takes.
136
+ # Everything else is absent, and the absence is the same refusal ClickHouse makes for `bytes`:
137
+ # a neutral type with no entry here cannot be given a column, so the group cannot be placed.
138
+ # It matters less here than there, because a model with an extra *field* is already refused by
139
+ # `fixed_schema_mismatch` - but the two refusals have to agree, and the cheapest way to make
140
+ # them agree is for this table to hold exactly the types the shape names.
141
+ "string": "char*",
142
+ "int32": "uint32_t",
143
+ "int64": "int64_t",
144
+ }
145
+
146
+ _DIALECT_TYPES: Final[Mapping[str, Mapping[str, str]]] = {
147
+ "postgres": POSTGRES_TYPES,
148
+ "clickhouse": CLICKHOUSE_TYPES,
149
+ "orderbook": ORDERBOOK_TYPES,
150
+ }
151
+
152
+ # The dialects this library knows, as one public list. It exists because the control plane kept its
153
+ # own vocabulary - `postgresql` there, `postgres` here - and the two agreed only for as long as
154
+ # nothing joined them. The first code that needed both spellings to match was the one rendering DDL
155
+ # for an engine named in the registry, and it failed at runtime rather than at any earlier point.
156
+ DIALECTS: Final[tuple[str, ...]] = ("clickhouse", "orderbook", "postgres")
157
+
158
+ _DECIMAL: Final[Mapping[str, str]] = {
159
+ "postgres": "numeric({digits},{scale})",
160
+ "clickhouse": "Decimal({digits}, {scale})",
161
+ }
162
+
163
+
164
+ # ── An engine whose schema is not ours to choose ─────────────────────────────────────────────────
165
+ #
166
+ # Everything above answers "what columns should this entity get here". The orderbook engine answers
167
+ # it first: it stores L2 depth in one shape, fixed in C++, and there is no CREATE TABLE to send it.
168
+ # So the relationship inverts. For PostgreSQL and ClickHouse the client declares a model and we
169
+ # decide the physical schema; here the engine has already decided, and either the client's model
170
+ # *is* that shape or the group cannot go there.
171
+ #
172
+ # That is stated rather than smoothed over, because smoothing it over has only bad forms. Mapping
173
+ # the client's field names onto the engine's - `at` onto `timestamp_ns`, `qty` onto `quantity` -
174
+ # would mean guessing which declared field is the price from what it is called, and reasoning from a
175
+ # name is the one thing this product refuses everywhere else. Accepting a wider model and dropping
176
+ # the extra fields would lose data in an engine chosen for not losing any.
177
+ #
178
+ # The consequence is a constraint on the model, and the refusal names the whole expected shape so it
179
+ # is actionable in one read.
180
+
181
+ ORDERBOOK_TABLE: Final[str] = "orderbook"
182
+ """The engine's own name for its storage. One table, and we did not name it."""
183
+
184
+ ORDERBOOK_SHAPE: Final[Mapping[str, str]] = {
185
+ # The address. Not columns in a result row - the engine's query language takes them in the FROM
186
+ # clause, `FROM 'BTCUSDT'.'binance'` - but they are fields of the entity and part of its key,
187
+ # because two rows differing only in symbol are different rows.
188
+ "symbol": "string",
189
+ "exchange": "string",
190
+ # The row.
191
+ "timestamp_ns": "int64",
192
+ "side": "string",
193
+ "level": "int32",
194
+ "price": "int64",
195
+ "quantity": "int64",
196
+ "order_count": "int32",
197
+ # Nullable in the model and *unknown* rather than zero when the engine cannot supply it. The
198
+ # engine returns 0 for a row whose sequence number it does not have, which is a safe sentinel
199
+ # there because its own numbering starts at 1 - and it is not safe here, because a client
200
+ # comparing sequence numbers cannot tell a sentinel from a value. The adapter converts.
201
+ "sequence_number": "int64",
202
+ }
203
+ """The nine fields an entity must declare, by name and by neutral type, to live in this engine.
204
+
205
+ ``price`` and ``quantity`` are integers in the engine's sub-unit, not decimals. That is the engine's
206
+ choice and it is the right one for an orderbook - a decimal per level per update is a rounding
207
+ question in the hot path - but it means a client who declared ``decimal(12,2)`` is declaring a
208
+ different model, and this refuses rather than converting. Converting would put our arithmetic
209
+ between the client's price and their storage.
210
+ """
211
+
212
+ ORDERBOOK_KEY: Final[tuple[str, ...]] = (
213
+ "symbol",
214
+ "exchange",
215
+ "timestamp_ns",
216
+ "side",
217
+ "level",
218
+ )
219
+ """The key, in the order the engine addresses by. Positional and load-bearing, as in ClickHouse."""
220
+
221
+ FIXED_SCHEMA: Final[frozenset[str]] = frozenset({"orderbook"})
222
+ """Dialects whose physical schema the engine imposes rather than accepting from us.
223
+
224
+ A named set rather than a check on the dialect string, so that the three places that have to behave
225
+ differently - the layout, the DDL renderer and the control plane's eligibility filter - agree by
226
+ construction instead of each testing for one name.
227
+ """
228
+
229
+ _FIXED_SHAPES: Final[Mapping[str, tuple[str, Mapping[str, str], tuple[str, ...]]]] = {
230
+ "orderbook": (ORDERBOOK_TABLE, ORDERBOOK_SHAPE, ORDERBOOK_KEY),
231
+ }
232
+
233
+ assert set(_FIXED_SHAPES) == set(FIXED_SCHEMA), sorted(set(_FIXED_SHAPES) ^ set(FIXED_SCHEMA))
234
+
235
+
236
+ def fixed_schema_mismatch(
237
+ columns: Mapping[str, Mapping[str, str]], *, dialect: str
238
+ ) -> str | None:
239
+ """Why a group cannot live in a fixed-schema engine, or ``None`` if it can.
240
+
241
+ ``columns`` is entity name to column name to neutral type - what :func:`group_columns` returns.
242
+ ``None`` for a dialect that is not fixed-schema, because "no objection" is the honest answer to
243
+ a question that does not apply: the caller asks this once per engine and branching on the
244
+ dialect at the call site would put the set of fixed-schema engines in two places.
245
+
246
+ A type check alone is not enough and that is the whole reason this exists. The orderbook shape
247
+ is made of ``string``, ``int32`` and ``int64``, all of which every engine can store - so a model
248
+ of two integers and a string passes representability and then fails while the layout is built,
249
+ which is the failure this function was added to stop happening twice.
250
+ """
251
+ fixed = _FIXED_SHAPES.get(dialect)
252
+ if fixed is None:
253
+ return None
254
+ _, shape, _ = fixed
255
+
256
+ if len(columns) != 1:
257
+ return (
258
+ f"this engine stores one thing, and this group has {len(columns)} entities "
259
+ f"({sorted(columns)}). A colocation group is what shares an engine, so a group of two "
260
+ f"cannot go somewhere with room for one."
261
+ )
262
+ entity, declared = next(iter(columns.items()))
263
+
264
+ missing = sorted(set(shape) - set(declared))
265
+ extra = sorted(set(declared) - set(shape))
266
+ wrong = sorted(
267
+ f"{name} is declared {declared[name]!r} and this engine stores {shape[name]!r}"
268
+ for name in sorted(set(shape) & set(declared))
269
+ if declared[name] != shape[name]
270
+ )
271
+ if not (missing or extra or wrong):
272
+ return None
273
+
274
+ problems = []
275
+ if missing:
276
+ problems.append(f"{entity} declares no {missing}")
277
+ if extra:
278
+ problems.append(f"{entity} declares {extra}, which this engine has nowhere to put")
279
+ if wrong:
280
+ problems.append("; ".join(wrong))
281
+ expected = ", ".join(f"{name}: {kind}" for name, kind in shape.items())
282
+ return (
283
+ f"{'. '.join(problems)}. This engine's schema is fixed in the engine and not chosen by us, "
284
+ f"so a model either is that shape or cannot be stored here. The shape is exactly: "
285
+ f"{expected}."
286
+ )
287
+
288
+
289
+ def _column_type(neutral: str, dialect: str) -> str:
290
+ """One neutral type, one engine type, and no defaulting.
291
+
292
+ A missing entry raises rather than falling back to a string, because a type nobody mapped is a
293
+ gap in an adapter and the honest place to find that out is here - not in a client's database,
294
+ where the column already exists with the wrong type and changing it is a migration.
295
+ """
296
+ types = _DIALECT_TYPES[dialect]
297
+ if neutral.startswith("decimal("):
298
+ digits, scale = neutral[len("decimal(") : -1].split(",")
299
+ template = _DECIMAL.get(dialect)
300
+ if template is None:
301
+ # A DeclarationError rather than a KeyError, because this is the same answer as any
302
+ # other unmapped neutral type - the engine cannot give this column a type - and
303
+ # `can_store` distinguishes "cannot store" from "does not parse" by which exception it
304
+ # sees. The orderbook engine stores prices as integers in a sub-unit; a decimal there
305
+ # would need our arithmetic between the client's price and their storage.
306
+ raise DeclarationError(
307
+ f"no {dialect} type for {neutral!r}. This engine has no decimal type at all, so "
308
+ f"this is a gap in the model for this engine rather than in the adapter."
309
+ )
310
+ return template.format(digits=digits, scale=scale)
311
+ try:
312
+ return types[neutral]
313
+ except KeyError:
314
+ raise DeclarationError(
315
+ f"no {dialect} type for {neutral!r}. Every member of the neutral vocabulary needs one; "
316
+ "this is a gap in the adapter rather than a problem with the model."
317
+ ) from None
318
+
319
+
320
+ def can_store(neutral: str, *, dialect: str) -> bool:
321
+ """Whether this dialect has a column type for this neutral type.
322
+
323
+ The one public question about representability, and it exists because the answer was previously
324
+ only obtainable by trying: a group with a ``bytes`` field placed in ClickHouse produced a
325
+ valid scoring decision and then raised out of ``default_layout``. The refusal was correct and it
326
+ landed in the wrong place - at map-build time, where it reads as our defect rather than as a
327
+ reason one engine was not a candidate.
328
+
329
+ Implemented by asking the renderer rather than by a second table. A representability check that
330
+ could disagree with the type mapping would be worse than none: it would let a placement be
331
+ approved and then fail to apply, which is the failure the byte contract exists to prevent.
332
+
333
+ A malformed type - ``decimal(x)`` - is not answered ``False``. "This engine cannot store it"
334
+ is a claim about the engine; a type that parses nowhere is a claim about the model, and the
335
+ model's own validation owns it. So that raises through.
336
+ """
337
+ if dialect not in _DIALECT_TYPES:
338
+ raise DeclarationError(
339
+ f"no type table for dialect {dialect!r}; this library knows {sorted(_DIALECT_TYPES)}. "
340
+ f"Answering False would say 'that engine cannot store it' about an engine this library "
341
+ f"has never heard of."
342
+ )
343
+ try:
344
+ _column_type(neutral, dialect)
345
+ except DeclarationError:
346
+ return False
347
+ return True
348
+
349
+
350
+ def _neutral_columns(model: LogicalModel, group: Group) -> dict[str, dict[str, str]]:
351
+ """Every column a group's tables need, in the neutral vocabulary, before any dialect.
352
+
353
+ Factored out of ``default_layout`` because two things need it and a second copy of the loop is
354
+ how they stop agreeing. ``default_layout`` maps each of these through ``_column_type``;
355
+ ``stored_types`` asks which of them an engine can represent at all.
356
+
357
+ A group has a foreign-key column for every relation whose source is a member. Today those add no
358
+ *type* the group did not already have, because a relation unions its two ends into one
359
+ colocation group, so the target's key fields are declared fields of a member. That is a fact
360
+ about how groups are formed rather than about layouts, and this function does not rely on it -
361
+ which is the point of deriving the set here instead of from ``spec.fields`` at the call site.
362
+
363
+ Insertion order is the declared field order, then relations in declared order. Callers that need
364
+ a stable ordering get one without sorting; callers that sort - the DDL renderer does - are
365
+ unaffected.
366
+ """
367
+ out: dict[str, dict[str, str]] = {}
368
+ for entity_name in group.members:
369
+ spec = model.entity(entity_name)
370
+ cols: dict[str, str] = {field.name: field.type for field in spec.fields}
371
+ for relation in model.relations:
372
+ if relation.source != entity_name:
373
+ continue
374
+ target = model.entity(relation.target)
375
+ for key_field in target.key:
376
+ key_spec = next(f for f in target.fields if f.name == key_field)
377
+ # <relation>_<key field>, for a single-field key and for a composite one alike. An
378
+ # earlier version branched on the key's arity and produced the same name in both
379
+ # arms, which ruff spotted as a useless condition - correctly, and it was a leftover
380
+ # from an idea about naming that turned out not to be worth the inconsistency.
381
+ cols[f"{relation.name}_{key_field}"] = key_spec.type
382
+ out[entity_name] = cols
383
+ return out
384
+
385
+
386
+ def group_columns(model: LogicalModel, group: Group) -> Mapping[str, Mapping[str, str]]:
387
+ """Every column a group's tables need, per entity, in the neutral vocabulary.
388
+
389
+ The public form of the derivation the layout uses. Exposed because the control plane has two
390
+ questions to ask before it picks an engine - can this engine store these types, and does this
391
+ group *fit* an engine whose shape is fixed - and the second one needs the column names, not just
392
+ the types.
393
+ """
394
+ return _neutral_columns(model, group)
395
+
396
+
397
+ def stored_types(model: LogicalModel, group: Group) -> tuple[str, ...]:
398
+ """The neutral types a group's tables need, sorted and deduplicated.
399
+
400
+ For the control plane, which has to answer "can this engine hold this group" *before* it picks
401
+ one. Without this the only available answer was to place the group and watch ``default_layout``
402
+ raise - a correct refusal in the wrong place, because it arrives after the decision is made and
403
+ reads as our defect rather than as the reason an engine was not a candidate.
404
+
405
+ Paired with :func:`can_store`, which is asked once per type. Both come from the same column
406
+ derivation the layout uses, so an engine reported able to hold a group is one whose layout can
407
+ actually be rendered - and that equivalence is what the control plane relies on when it excludes
408
+ an engine instead of discovering the problem while building the map.
409
+ """
410
+ return tuple(
411
+ sorted(
412
+ {
413
+ column_type
414
+ for cols in _neutral_columns(model, group).values()
415
+ for column_type in cols.values()
416
+ }
417
+ )
418
+ )
419
+
420
+
421
+ def default_layout(
422
+ model: LogicalModel, group: Group, *, dialect: str = "postgres"
423
+ ) -> PhysicalLayout:
424
+ """The obvious schema for a group in one engine.
425
+
426
+ Foreign key columns are named ``<relation>_<target key field>``, which is the convention nearly
427
+ every ORM uses, so a client looking at their own database recognises what they are seeing. A
428
+ relation to an entity with a composite key produces one column per key field.
429
+ """
430
+ if dialect not in _DIALECT_TYPES:
431
+ raise DeclarationError(
432
+ f"no default layout for dialect {dialect!r}. Adding one is an adapter's job, and it "
433
+ f"has "
434
+ f"to be added deliberately rather than approximated from an existing one - the known "
435
+ f"dialects are {sorted(_DIALECT_TYPES)}."
436
+ )
437
+
438
+ neutral = _neutral_columns(model, group)
439
+
440
+ if dialect in FIXED_SCHEMA:
441
+ mismatch = fixed_schema_mismatch(neutral, dialect=dialect)
442
+ if mismatch is not None:
443
+ raise DeclarationError(mismatch)
444
+ table, shape, _ = _FIXED_SHAPES[dialect]
445
+ entity = next(iter(neutral))
446
+ # The engine's own table name and the engine's own column names, typed through the same
447
+ # `_column_type` as everything else so that an unmapped neutral type still raises here
448
+ # rather than producing a column with a plausible C type nobody chose.
449
+ return PhysicalLayout(
450
+ tables={entity: table},
451
+ columns={entity: {name: _column_type(kind, dialect) for name, kind in shape.items()}},
452
+ indexes=(),
453
+ )
454
+
455
+ tables: dict[str, str] = {}
456
+ columns: dict[str, dict[str, str]] = {}
457
+ indexes: list[dict[str, object]] = []
458
+
459
+ for entity_name in group.members:
460
+ tables[entity_name] = snake_case(entity_name)
461
+ columns[entity_name] = {
462
+ column: _column_type(neutral_type, dialect)
463
+ for column, neutral_type in neutral[entity_name].items()
464
+ }
465
+
466
+ # One index, and only because a foreign key without one turns every relation walk into a
467
+ # sequential scan. Anything beyond this is a planner decision with a cost attached, and the
468
+ # library has no business guessing at it.
469
+ #
470
+ # None at all for ClickHouse, and that is not an omission. It has no B-tree: `CREATE INDEX`
471
+ # there builds a *data-skipping* index, which needs a type and a granularity and helps only
472
+ # when the data is already ordered so whole granules can be ruled out. Choosing those is a
473
+ # planner decision with a cost attached, exactly like the ones above that this function
474
+ # refuses to make. What a MergeTree does have is its `ORDER BY`, which is the primary index
475
+ # and is derived from the declared key by the adapter - so the useful index exists, it is
476
+ # simply not expressed as a row in this list.
477
+ if dialect != "clickhouse":
478
+ for relation in model.relations:
479
+ if relation.source != entity_name:
480
+ continue
481
+ target = model.entity(relation.target)
482
+ indexes.append(
483
+ {
484
+ "entity": entity_name,
485
+ "name": f"{snake_case(entity_name)}_{relation.name}_idx",
486
+ "columns": [f"{relation.name}_{k}" for k in target.key],
487
+ }
488
+ )
489
+
490
+ return PhysicalLayout(tables=tables, columns=columns, indexes=tuple(indexes))
491
+
492
+
493
+ # ── The other layout: one wide table for reading, and the field it will not copy ─────────────────
494
+
495
+
496
+ @dataclass(frozen=True)
497
+ class DerivedLayout:
498
+ """A denormalised layout, plus what it left behind and why.
499
+
500
+ One value rather than a layout and a second function answering "what was excluded", because two
501
+ derivations of the same decision drift - and the thing that would drift here is a list of
502
+ personal data fields, where a stale answer is the worst kind. The control plane puts ``layout``
503
+ in the map and ``excluded_pii`` in the report; both come from one pass.
504
+ """
505
+
506
+ layout: PhysicalLayout
507
+ root: str
508
+ """The entity whose grain the wide table has: one row per root row."""
509
+
510
+ inlined: tuple[tuple[str, str], ...]
511
+ """(relation, target entity) pairs that were flattened in, in declared order."""
512
+
513
+ excluded_pii: tuple[str, ...]
514
+ """``Entity.field`` for every personal-data field kept out, sorted.
515
+
516
+ Reported rather than silently absent. An analyst who queries the wide table and finds no email
517
+ needs to be told that the column is missing *by decision*, not conclude the data is incomplete -
518
+ and the client needs to see which fields they would have to allow explicitly.
519
+ """
520
+
521
+
522
+ def _relation_columns(model: LogicalModel, group: Group) -> dict[str, list[Any]]:
523
+ """Intra-group relations by source entity, in declared order."""
524
+ out: dict[str, list[Any]] = {name: [] for name in group.members}
525
+ members = set(group.members)
526
+ for relation in model.relations:
527
+ if relation.source in members and relation.target in members:
528
+ out[relation.source].append(relation)
529
+ return out
530
+
531
+
532
+ def denormalized_layout(
533
+ model: LogicalModel,
534
+ group: Group,
535
+ *,
536
+ dialect: str = "clickhouse",
537
+ include_pii: Sequence[str] = (),
538
+ ) -> DerivedLayout:
539
+ """One wide table for a group, for an engine that is read from rather than written to.
540
+
541
+ Requirement 5.1: the same entity may be materialised differently in different engines -
542
+ normalised in a transactional store to write, flattened in an analytical one to read - from one
543
+ logical model. This is the second derivation, and it is here rather than in the control plane
544
+ for the same reason ``default_layout`` is: the library has to be able to apply a hand-written
545
+ map, and a schema derived in two places is a schema that eventually differs in one.
546
+
547
+ **Personal data is excluded unless named in ``include_pii``.** Requirement 5.5, and it lives in
548
+ this function rather than in the planner on purpose: an analytical materialisation is a second
549
+ copy of the data in a different engine with different access controls, so "we do not copy your
550
+ personal data there" is a promise worth being able to *read* rather than take. Fails closed - an
551
+ empty allowance excludes every declared personal-data field, and a caller who wants one copied
552
+ names it. Names are ``Entity.field``, because a group holds several entities and ``email`` alone
553
+ would be ambiguous the moment two of them have one.
554
+
555
+ The rule applies to the root entity's own fields as well as to the inlined ones. "Denormalised
556
+ into an analytical materialisation" is about the copy arriving there, not about which side of a
557
+ join it came from, and a rule with an exception for the root would be a rule nobody could state
558
+ in one sentence.
559
+
560
+ Four refusals, each because the friendly alternative produces something worse than an error:
561
+
562
+ **A group with no intra-group relation.** There is nothing to flatten, so the wide table would
563
+ be ``default_layout`` under another name: a second copy costing storage, replication and lag
564
+ while answering exactly the questions the source already answers.
565
+
566
+ **An ambiguous root.** The grain is the entity nothing else in the group points at. If two
567
+ entities qualify, one row of the wide table would mean whatever this function guessed, and a
568
+ table whose row nobody chose the meaning of is worse than no table.
569
+
570
+ **A fixed-schema engine.** Its schema is not ours to choose, so there is no wide table to make
571
+ there - see ``FIXED_SCHEMA``.
572
+
573
+ **An allowance naming something that is not a declared personal-data field of this group.** Both
574
+ directions: a typo would leave the field excluded while the client believed they had allowed it,
575
+ and a name that is not personal data at all suggests the client thinks it is.
576
+ """
577
+ if dialect in FIXED_SCHEMA:
578
+ raise DeclarationError(
579
+ f"{dialect!r} imposes its own schema, so there is no denormalised layout to derive for "
580
+ f"it. A wide table is a choice about physical shape, and this engine has already made "
581
+ f"it."
582
+ )
583
+
584
+ pii_by_entity = {
585
+ name: set(model.entity(name).pii) for name in group.members
586
+ }
587
+ known_pii = sorted(
588
+ f"{entity}.{field}" for entity, fields in pii_by_entity.items() for field in fields
589
+ )
590
+ allowed = tuple(include_pii)
591
+ unknown = sorted(set(allowed) - set(known_pii))
592
+ if unknown:
593
+ raise DeclarationError(
594
+ f"include_pii names {unknown}, which {group.name} does not declare as personal data. "
595
+ f"Refused in both directions rather than ignored: a misspelling would leave the field "
596
+ f"excluded while you believed you had allowed it, and a name that is not personal data "
597
+ f"suggests it was expected to be. This group declares {known_pii or 'none'}."
598
+ )
599
+
600
+ relations = _relation_columns(model, group)
601
+ targets = {r.target for outgoing in relations.values() for r in outgoing}
602
+ roots = sorted(name for name in group.members if name not in targets)
603
+ if not any(relations.values()):
604
+ raise DeclarationError(
605
+ f"{group.name} has no relation inside it, so there is nothing to denormalise. A wide "
606
+ f"table here would be the default layout under another name - a second copy paying for "
607
+ f"storage, replication and lag to answer the questions the source already answers."
608
+ )
609
+ if len(roots) != 1:
610
+ raise DeclarationError(
611
+ f"{group.name} has {roots} as candidate roots and a wide table has one grain. Refusing "
612
+ f"rather than picking: one row would mean whatever this function chose, and a table "
613
+ f"whose row nobody decided the meaning of is worse than no table."
614
+ )
615
+ root = roots[0]
616
+
617
+ neutral = _neutral_columns(model, group)
618
+ columns: dict[str, str] = {}
619
+ excluded: list[str] = []
620
+
621
+ def take(entity: str, field: str, column: str, neutral_type: str) -> None:
622
+ qualified = f"{entity}.{field}"
623
+ if field in pii_by_entity[entity] and qualified not in allowed:
624
+ excluded.append(qualified)
625
+ return
626
+ columns[column] = _column_type(neutral_type, dialect)
627
+
628
+ # The root's own columns keep their names. A column belonging to an inlined relation is skipped
629
+ # here and taken on the inlined pass instead, and that is **not** about duplicates: the foreign
630
+ # key `user_id` and the inlined `User.id` produce the same column name by construction, so the
631
+ # dictionary assignment would be idempotent either way. It is about the personal-data rule. A
632
+ # foreign-key column is named after the relation, not after a field of the root, so the check in
633
+ # `take()` does not recognise it - and a target whose *key* is declared personal data
634
+ # (`Person.national_id`) would arrive here under `person_national_id` with nothing to stop it.
635
+ # Deferring to the inlined pass puts the column under the entity that declared it.
636
+ inlined_names = {relation.name for relation in relations[root]}
637
+ for column, neutral_type in neutral[root].items():
638
+ if any(column.startswith(f"{name}_") for name in inlined_names):
639
+ continue
640
+ take(root, column, column, neutral_type)
641
+
642
+ inlined: list[tuple[str, str]] = []
643
+ for relation in relations[root]:
644
+ target = relation.target
645
+ inlined.append((relation.name, target))
646
+ for column, neutral_type in neutral[target].items():
647
+ # `<relation>_<field>`, the same convention the foreign key already uses, so a client
648
+ # reading their own analytical table recognises where a column came from without a map.
649
+ take(target, column, f"{relation.name}_{column}", neutral_type)
650
+
651
+ return DerivedLayout(
652
+ layout=PhysicalLayout(
653
+ tables={root: f"{snake_case(root)}_wide"},
654
+ columns={root: columns},
655
+ indexes=(),
656
+ ),
657
+ root=root,
658
+ inlined=tuple(inlined),
659
+ excluded_pii=tuple(sorted(excluded)),
660
+ )