smart-data-engine-sdk 0.1.0.dev0__py3-none-any.whl

This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
@@ -0,0 +1,14 @@
1
+ """Support for running the shared conformance vectors, and for testing an adapter.
2
+
3
+ Two modules, both public and both narrow. :mod:`sde.testing.loader` turns a vector's neutral
4
+ declaration into a model, which every implementation needs because a vector cannot contain one
5
+ language's classes. :mod:`sde.testing.memory` is an in-memory engine, which every implementation
6
+ needs because the ``migration/`` vectors pin behaviour that only happens against one.
7
+ """
8
+
9
+ from __future__ import annotations
10
+
11
+ from .loader import model_from_neutral
12
+ from .memory import MemoryEngine, Recorded, engines_from
13
+
14
+ __all__ = ["MemoryEngine", "Recorded", "engines_from", "model_from_neutral"]
sde/testing/loader.py ADDED
@@ -0,0 +1,175 @@
1
+ """Building a model from neutral JSON, so that conformance vectors can be shared.
2
+
3
+ Every library needs this. A vector cannot contain Python decorators or TypeScript classes, so the
4
+ declaration in a vector is plain JSON and each implementation needs a way to turn that JSON into
5
+ whatever its own model type is. It is a small amount of code and it is a requirement, not a
6
+ convenience: without it the vectors could not be common, and without common vectors four
7
+ implementations drift apart silently.
8
+
9
+ It deliberately does *not* re-implement the encoding. It produces the same specs the decorator path
10
+ produces and hands them to :func:`sde.model.assemble`. A loader that built its own IR would make the
11
+ vectors verify a code path no application ever executes, which is the most expensive kind of green
12
+ test - it looks like coverage and is the absence of it.
13
+ """
14
+
15
+ from __future__ import annotations
16
+
17
+ from collections.abc import Mapping
18
+ from typing import Any
19
+
20
+ from ..errors import DeclarationError
21
+ from ..model import EntitySpec, FieldSpec, LogicalModel, RelationSpec, assemble
22
+ from ..types import NEUTRAL_TYPES
23
+
24
+ __all__ = ["model_from_neutral"]
25
+
26
+ _DECIMAL_PREFIX = "decimal("
27
+
28
+
29
+ def _check_type(neutral: str, where: str) -> str:
30
+ if neutral in NEUTRAL_TYPES:
31
+ return neutral
32
+
33
+ # Anything beginning with "decimal" gets the decimal-specific message, including a bare
34
+ # "decimal" with no parameters at all. A conformance vector caught this: the generic "not in the
35
+ # vocabulary" message is true and useless, because the reader's actual mistake is that they left
36
+ # off the precision, and telling them the word is unknown sends them looking for the right word.
37
+ if neutral == "decimal" or neutral.startswith("decimal"):
38
+ if neutral.startswith(_DECIMAL_PREFIX) and neutral.endswith(")"):
39
+ body = neutral[len(_DECIMAL_PREFIX) : -1]
40
+ parts = body.split(",")
41
+ if len(parts) == 2 and all(p.strip().isdigit() for p in parts):
42
+ digits, scale = (int(p) for p in parts)
43
+ if body == f"{digits},{scale}" and digits >= 1 and 0 <= scale <= digits:
44
+ return neutral
45
+ raise DeclarationError(
46
+ f"{where}: {neutral!r} is not a well-formed decimal. The written form is "
47
+ "decimal(digits,scale) - precision then scale, no spaces, both required. No spaces "
48
+ "because whitespace inside a type name is exactly the sort of thing two libraries "
49
+ "would "
50
+ "disagree about, and both required because a decimal without precision is not a "
51
+ "storable type in any engine we place data in."
52
+ )
53
+
54
+ raise DeclarationError(
55
+ f"{where}: {neutral!r} is not in the neutral type vocabulary "
56
+ f"({', '.join(sorted(NEUTRAL_TYPES))}, decimal(p,s))"
57
+ )
58
+
59
+
60
+ def _check_shape(raw: Mapping[str, Any], name: str) -> None:
61
+ """The shapes a person can plausibly hand this function, named rather than crashed on.
62
+
63
+ The first one is the whole reason this exists. The IR and the neutral form are close enough to
64
+ be confused - our own tests declared a model's IR to the control plane and got away with it for
65
+ weeks, because nothing read the stored document back - and they differ exactly where a key is
66
+ written. Handing the IR over used to reach a set membership test on a dict and raise
67
+ ``TypeError: unhashable type: 'dict'`` from inside the encoder, which tells the reader nothing
68
+ about the mistake they made. Vector ``errors/036``.
69
+ """
70
+ fields = raw.get("fields")
71
+ if not isinstance(fields, list):
72
+ raise DeclarationError(f"{name}: 'fields' is a list, and this one is {fields!r}")
73
+ for field in fields:
74
+ if not isinstance(field, Mapping) or not isinstance(field.get("name"), str):
75
+ raise DeclarationError(f"{name}: a field is an object with a name, not {field!r}")
76
+ if not isinstance(field.get("type"), str):
77
+ raise DeclarationError(f"{name}.{field['name']}: a field needs a type")
78
+
79
+ key = raw.get("key")
80
+ if key is not None and not isinstance(key, list):
81
+ raise DeclarationError(f"{name}: 'key' is a list of field names, not {key!r}")
82
+ for part in key or ():
83
+ if isinstance(part, Mapping) and "field" in part and "position" in part:
84
+ raise DeclarationError(
85
+ f"{name}: 'key' holds {part!r}, which is the IR's key form rather than the neutral "
86
+ "one. The IR records a position because array order is not load-bearing anywhere "
87
+ "else in it; a declaration states a key as a list of field names, in order. If you "
88
+ "meant to hand over a model you already built, sde.neutral_declaration() produces "
89
+ "this document from it."
90
+ )
91
+ if not isinstance(part, str):
92
+ raise DeclarationError(f"{name}: 'key' names fields as strings, not {part!r}")
93
+
94
+ for plural in ("pii",):
95
+ value = raw.get(plural)
96
+ if value is not None and (
97
+ not isinstance(value, list) or any(not isinstance(v, str) for v in value)
98
+ ):
99
+ raise DeclarationError(f"{name}: {plural!r} is a list of field names, not {value!r}")
100
+
101
+
102
+ def model_from_neutral(data: Mapping[str, Any]) -> LogicalModel:
103
+ """Build a :class:`~sde.model.LogicalModel` from a vector's ``model.json``.
104
+
105
+ The neutral form states keys as a plain list, because that is what a human writes. Turning it
106
+ into the positioned form the IR uses is this library's job, which is the point: if the vector
107
+ carried the positioned form we would be checking that we can copy JSON.
108
+
109
+ The shape of the document is checked before anything is read out of it. Not defensiveness:
110
+ this is the entry point a client's own tooling hits and the one a new implementation writes
111
+ first, and a bare ``KeyError`` out of a dictionary lookup is the failure mode section 7 of the
112
+ format contract already names for a missing rule - a library exception with no explanation, on
113
+ the path whose whole job is to explain.
114
+ """
115
+ if not isinstance(data, Mapping):
116
+ raise DeclarationError("a neutral model declaration is an object with an 'entities' list")
117
+ raw_entities = data.get("entities")
118
+ if not isinstance(raw_entities, list):
119
+ raise DeclarationError("a neutral model declaration needs an 'entities' list")
120
+
121
+ entities: list[EntitySpec] = []
122
+ for raw in raw_entities:
123
+ if not isinstance(raw, Mapping):
124
+ raise DeclarationError(f"an entity is an object, and this one is {raw!r}")
125
+ if not isinstance(raw.get("name"), str) or not raw["name"]:
126
+ raise DeclarationError(f"an entity needs a name, and this one has {raw.get('name')!r}")
127
+ name = raw["name"]
128
+ _check_shape(raw, name)
129
+ fields = tuple(
130
+ FieldSpec(
131
+ name=f["name"],
132
+ type=_check_type(f["type"], f"{name}.{f['name']}"),
133
+ nullable=bool(f.get("nullable", False)),
134
+ )
135
+ for f in raw.get("fields", ())
136
+ )
137
+ # No default for the key. This line used to read `raw.get("key") or ("id",)` and the
138
+ # TypeScript port spelled the same thing with `??`, which differs on exactly one input:
139
+ # `"key": []` invented a key here and stayed keyless there, so one declaration had two
140
+ # model versions and no vector could see it. The rule is format-contract §4a now, and every
141
+ # refusal about a declaration lives in `assemble`, where both front doors meet - this one
142
+ # enforced a subset and the vectors therefore ran a weaker validator than any application.
143
+ key = tuple(raw.get("key") or ())
144
+ entities.append(
145
+ EntitySpec(
146
+ name=name,
147
+ fields=tuple(sorted(fields, key=lambda f: f.name)),
148
+ key=key,
149
+ pii=tuple(sorted(raw.get("pii") or ())),
150
+ residency=raw.get("residency"),
151
+ )
152
+ )
153
+
154
+ names = {e.name for e in entities}
155
+ relations: list[RelationSpec] = []
156
+ for raw in data.get("relations", ()):
157
+ source, target = raw["from"], raw["to"]
158
+ for side in (source, target):
159
+ if side not in names:
160
+ raise DeclarationError(f"relation {raw['name']!r} names unknown entity {side!r}")
161
+ relations.append(RelationSpec(name=raw["name"], source=source, target=target))
162
+
163
+ atomic_raw = data.get("atomic") or ()
164
+ atomic = tuple(sorted(tuple(sorted(group)) for group in atomic_raw))
165
+ for group in atomic:
166
+ unknown = [m for m in group if m not in names]
167
+ if unknown:
168
+ raise DeclarationError(f"atomic group names unknown entities {unknown}")
169
+
170
+ return assemble(
171
+ entities=tuple(entities),
172
+ relations=tuple(relations),
173
+ atomic=atomic,
174
+ cost_ceiling=data.get("cost_ceiling"),
175
+ )
sde/testing/memory.py ADDED
@@ -0,0 +1,318 @@
1
+ """An in-memory engine, for the ``migration/`` conformance vectors and for anybody's adapter tests.
2
+
3
+ **Why this is in the library rather than in a test file.** The ``migration/`` vectors pin behaviour
4
+ that only happens against an engine: the order of the calls a backfill makes, the arithmetic of the
5
+ resume marker, which side of the marker a verify counter lands on. Every implementation therefore
6
+ needs an engine to run them against, and the two options were for each runner to write its own or
7
+ for the fixture to be shared. A runner that writes its own is a runner whose *fixture* can be the
8
+ thing that differs, and then a red vector means "one of two tables disagreed" rather than "one of
9
+ two libraries disagreed" - which is the failure the whole suite exists to avoid, one level down.
10
+
11
+ **What it is not.** It is not an implementation of anything in the format contract. It stores rows
12
+ in a list and answers questions about them; every rule the vectors check lives in
13
+ :mod:`sde.migration` and in :mod:`sde.watermark`. The one property it does have to get right is a
14
+ keyset scan, and the vectors pin the *calls* as well as the results, so a fixture that scanned
15
+ differently would show up as a different call sequence rather than as a plausible wrong answer.
16
+
17
+ It is also genuinely useful outside the vectors, which is why it is public: an adapter written
18
+ against :class:`sde.migration.Migratable` has the same behaviour to check, and checking it against a
19
+ real server is slow and checking it against nothing is what leaves a backfill copying the same chunk
20
+ forever.
21
+ """
22
+
23
+ from __future__ import annotations
24
+
25
+ from collections.abc import Iterator, Mapping, Sequence
26
+ from contextlib import contextmanager
27
+ from typing import Any
28
+
29
+ from ..errors import EngineError
30
+ from ..placement import PhysicalLayout
31
+
32
+ __all__ = ["MemoryEngine", "Recorded", "engines_from"]
33
+
34
+
35
+ class Recorded:
36
+ """The calls a set of engines received, in one sequence, each entry naming its engine.
37
+
38
+ Kept because the ``migration/`` vectors pin the sequence and not only the outcome. A library
39
+ that reached the same progress record by scanning the whole table and filtering in memory would
40
+ satisfy every count and be unusable on a real table; the calls are the part that says how the
41
+ result was obtained.
42
+
43
+ **One journal for the whole engine set, and that is a fix.** The first version kept a list per
44
+ engine, which cannot express the guarantee the dual-write cases are about: a row reaches the
45
+ source *before* anything is attempted against the copy. Reversing those two lines passed every
46
+ vector, because each engine's own list was still in order - and the vector's own note claimed
47
+ that ordering was what it pinned. A guarantee across two engines needs one sequence.
48
+ """
49
+
50
+ def __init__(self) -> None:
51
+ self.calls: list[dict[str, Any]] = []
52
+
53
+ def note(self, engine: str, method: str, **arguments: Any) -> None:
54
+ self.calls.append({"engine": engine, "call": method, **arguments})
55
+
56
+ def as_list(self) -> list[dict[str, Any]]:
57
+ return [dict(entry) for entry in self.calls]
58
+
59
+
60
+ def _sortable(value: Any) -> tuple[int, Any]:
61
+ """A total order over the value kinds a vector may use, with the kind first.
62
+
63
+ Comparing a string to an integer raises in Python and coerces in JavaScript, and neither is a
64
+ key order. Vectors use one kind per column, so this never has to decide *between* kinds for a
65
+ real comparison - the tag is there so that a vector which accidentally mixed them fails loudly
66
+ in both languages instead of one.
67
+ """
68
+ if value is None:
69
+ return (0, 0)
70
+ if isinstance(value, bool):
71
+ return (1, int(value))
72
+ if isinstance(value, (int, float)):
73
+ return (2, value)
74
+ return (3, str(value))
75
+
76
+
77
+ def _key(order: Sequence[str], row: Mapping[str, Any]) -> tuple[tuple[int, Any], ...]:
78
+ return tuple(_sortable(row.get(column)) for column in order)
79
+
80
+
81
+ class MemoryEngine:
82
+ """One dialect, a set of named tables, and the two optional protocols an adapter may offer.
83
+
84
+ ``dialect`` matters: the migration gate refuses a copy between two dialects whose precision
85
+ differs, so a fixture that reported one dialect for both ends could not reach that refusal.
86
+
87
+ ``can_keep_bookkeeping`` and ``can_migrate`` exist so a vector can build an engine that
88
+ **cannot** take part, which is the case both of those refusals are about. They remove the
89
+ methods rather than making them fail, because that is what a real adapter without the capability
90
+ looks like - and the check is "will this object answer these calls".
91
+ """
92
+
93
+ def __init__(
94
+ self,
95
+ dialect: str = "postgres",
96
+ *,
97
+ name: str = "engine",
98
+ journal: Recorded | None = None,
99
+ tables: Mapping[str, Sequence[Mapping[str, Any]]] | None = None,
100
+ can_keep_bookkeeping: bool = True,
101
+ can_migrate: bool = True,
102
+ watermark: int | None = None,
103
+ markers: Mapping[tuple[str, str], int] | None = None,
104
+ fail_inserts: Mapping[str, int] | None = None,
105
+ ) -> None:
106
+ self.dialect = dialect
107
+ self.tables: dict[str, list[dict[str, Any]]] = {
108
+ name: [dict(row) for row in rows] for name, rows in (tables or {}).items()
109
+ }
110
+ self.name = name
111
+ self.recorded = journal if journal is not None else Recorded()
112
+ # How many of the next inserts into each table must fail. The one thing a fake has to be
113
+ # able to do that a real engine does on its own: a fan-out that does not reach the copy is
114
+ # the case the whole dual-write design is about, and it cannot be reached by writing
115
+ # correct rows to a working table.
116
+ self._fail_inserts: dict[str, int] = dict(fail_inserts or {})
117
+ self._watermarks: list[int] = [] if watermark is None else [watermark]
118
+ self._markers: dict[tuple[str, str], list[int]] = {
119
+ key: [value] for key, value in (markers or {}).items()
120
+ }
121
+ # **Bound onto the instance when the capability is on, and genuinely absent when it is
122
+ # off.** Setting them to ``None`` instead was the first attempt and it does not work:
123
+ # ``satisfies`` asks ``hasattr``, deliberately - a member that exists and is not callable
124
+ # fails at the call with a message naming it, which is a better failure than a capability
125
+ # check that quietly answers "no". So an attribute set to None reads as *present*, and the
126
+ # engine that was supposed to be unable to take part took part and crashed. Absence is also
127
+ # what a real adapter without the capability looks like.
128
+ if can_keep_bookkeeping:
129
+ self.map_watermark = self._map_watermark
130
+ self.record_map_version = self._record_map_version
131
+ if can_migrate:
132
+ self.key_range = self._key_range
133
+ self.nth_key = self._nth_key
134
+ self.copy_in = self._copy_in
135
+ self.count = self._count
136
+ self.backfill_marker = self._backfill_marker
137
+ self.record_backfill_marker = self._record_backfill_marker
138
+
139
+ # --- Engine ------------------------------------------------------------------------------
140
+
141
+ def ensure_schema(self, layout: PhysicalLayout, *, keys: Mapping[str, Any]) -> None:
142
+ self.recorded.note(self.name, "ensure_schema", tables=sorted(layout.tables.values()))
143
+ for table in layout.tables.values():
144
+ self.tables.setdefault(table, [])
145
+
146
+ def insert(self, table: str, values: Mapping[str, Any]) -> None:
147
+ self.recorded.note(self.name, "insert", table=table)
148
+ remaining = self._fail_inserts.get(table, 0)
149
+ if remaining > 0:
150
+ self._fail_inserts[table] = remaining - 1
151
+ raise EngineError(f"insert into {table} failed: this engine was told to refuse it")
152
+ self.tables.setdefault(table, []).append(dict(values))
153
+
154
+ def get(self, table: str, key: Mapping[str, Any]) -> dict[str, Any] | None:
155
+ self.recorded.note(self.name, "get", table=table)
156
+ for row in self.tables.get(table, []):
157
+ if all(row.get(column) == value for column, value in key.items()):
158
+ return dict(row)
159
+ return None
160
+
161
+ @contextmanager
162
+ def transaction(self) -> Iterator[MemoryEngine]:
163
+ """Snapshot, yield, and put the snapshot back on failure.
164
+
165
+ Enough to make a rollback observable, which is what the dual-write tests need: rows written
166
+ inside a transaction that raises must not reach the copy, and the only way to check that is
167
+ for the source to forget them too.
168
+ """
169
+ self.recorded.note(self.name, "transaction")
170
+ snapshot = {name: [dict(row) for row in rows] for name, rows in self.tables.items()}
171
+ try:
172
+ yield self
173
+ except BaseException:
174
+ self.tables = snapshot
175
+ raise
176
+
177
+ # --- WatermarkStore ----------------------------------------------------------------------
178
+
179
+ def _map_watermark(self) -> int | None:
180
+ self.recorded.note(self.name, "map_watermark")
181
+ return max(self._watermarks) if self._watermarks else None
182
+
183
+ def _record_map_version(self, version: int, *, model_version: str) -> None:
184
+ self.recorded.note(self.name, "record_map_version", version=version)
185
+ self._watermarks.append(version)
186
+
187
+ # --- Migratable --------------------------------------------------------------------------
188
+
189
+ def _key_range(
190
+ self,
191
+ table: str,
192
+ order: Sequence[str],
193
+ *,
194
+ after: Sequence[Any] | None = None,
195
+ upto: Sequence[Any] | None = None,
196
+ limit: int | None = None,
197
+ ) -> list[dict[str, Any]]:
198
+ from ..migration import key_columns, same_width
199
+
200
+ cols = key_columns(order, table)
201
+ if after is not None:
202
+ same_width(after, cols, "after")
203
+ if upto is not None:
204
+ same_width(upto, cols, "upto")
205
+ self.recorded.note(
206
+ self.name,
207
+ "key_range",
208
+ table=table,
209
+ after=None if after is None else list(after),
210
+ upto=None if upto is None else list(upto),
211
+ limit=limit,
212
+ )
213
+ rows = sorted(self.tables.get(table, []), key=lambda row: _key(cols, row))
214
+ if after is not None:
215
+ low = tuple(_sortable(value) for value in after)
216
+ rows = [row for row in rows if _key(cols, row) > low]
217
+ if upto is not None:
218
+ high = tuple(_sortable(value) for value in upto)
219
+ rows = [row for row in rows if _key(cols, row) <= high]
220
+ if limit is not None:
221
+ rows = rows[:limit]
222
+ return [dict(row) for row in rows]
223
+
224
+ def _nth_key(
225
+ self, table: str, order: Sequence[str], *, position: int
226
+ ) -> tuple[Any, ...] | None:
227
+ from ..migration import key_columns
228
+
229
+ cols = key_columns(order, table)
230
+ self.recorded.note(self.name, "nth_key", table=table, position=position)
231
+ if position < 1:
232
+ raise EngineError(f"position is one-based; {position} is not a row")
233
+ rows = sorted(self.tables.get(table, []), key=lambda row: _key(cols, row))
234
+ if position > len(rows):
235
+ return None
236
+ row = rows[position - 1]
237
+ return tuple(row.get(column) for column in cols)
238
+
239
+ def _copy_in(self, table: str, rows: Sequence[Mapping[str, Any]]) -> None:
240
+ self.recorded.note(self.name, "copy_in", table=table, rows=len(rows))
241
+ if not rows:
242
+ return
243
+ columns = sorted(rows[0])
244
+ for row in rows:
245
+ if sorted(row) != columns:
246
+ raise EngineError(
247
+ f"copy_in into {table} was given rows with different columns "
248
+ f"({columns} and {sorted(row)}). A chunk comes from one table, so this is a "
249
+ f"caller assembling it from two."
250
+ )
251
+ existing = self.tables.setdefault(table, [])
252
+ # Idempotent on the whole row's identity, which is what both real targets do by a different
253
+ # mechanism: ON CONFLICT DO NOTHING in PostgreSQL and a ReplacingMergeTree collapse in
254
+ # ClickHouse. Not the same mechanism, which is why the adapters have live tests in every
255
+ # direction and this only has to absorb a recopy.
256
+ for row in rows:
257
+ same = (
258
+ all(present.get(column) == row.get(column) for column in row)
259
+ for present in existing
260
+ )
261
+ if not any(same):
262
+ existing.append(dict(row))
263
+
264
+ def _count(self, table: str) -> int:
265
+ self.recorded.note(self.name, "count", table=table)
266
+ return len(self.tables.get(table, []))
267
+
268
+ def _backfill_marker(self, *, materialization: str, entity: str) -> int:
269
+ self.recorded.note(
270
+ self.name, "backfill_marker", materialization=materialization, entity=entity
271
+ )
272
+ seen = self._markers.get((materialization, entity))
273
+ return max(seen) if seen else 0
274
+
275
+ def _record_backfill_marker(self, *, materialization: str, entity: str, rows: int) -> None:
276
+ self.recorded.note(
277
+ self.name,
278
+ "record_backfill_marker",
279
+ materialization=materialization,
280
+ entity=entity,
281
+ rows=rows,
282
+ )
283
+ self._markers.setdefault((materialization, entity), []).append(rows)
284
+
285
+
286
+ def engines_from(
287
+ spec: Mapping[str, Mapping[str, Any]], journal: Recorded | None = None
288
+ ) -> dict[str, MemoryEngine]:
289
+ """Build the engine set a ``migration/`` case describes.
290
+
291
+ Here rather than in each runner for the same reason the engine itself is here: two runners that
292
+ each read this document their own way can disagree about the *fixture*, and then a red vector
293
+ says "one of two tables differed" instead of "one of two libraries differed". The generator uses
294
+ this too, so the document a vector carries is read by exactly one piece of code per language.
295
+ """
296
+ shared = journal if journal is not None else Recorded()
297
+ built: dict[str, MemoryEngine] = {}
298
+ for name, body in spec.items():
299
+ built[name] = MemoryEngine(
300
+ dialect=str(body.get("dialect", "postgres")),
301
+ name=name,
302
+ journal=shared,
303
+ tables={
304
+ table: [dict(row) for row in rows]
305
+ for table, rows in (body.get("tables") or {}).items()
306
+ },
307
+ can_keep_bookkeeping=bool(body.get("bookkeeping", True)),
308
+ can_migrate=bool(body.get("migratable", True)),
309
+ watermark=body.get("watermark"),
310
+ markers={
311
+ (key.split("|", 1)[0], key.split("|", 1)[1]): int(value)
312
+ for key, value in (body.get("markers") or {}).items()
313
+ },
314
+ fail_inserts={
315
+ table: int(count) for table, count in (body.get("fail_inserts") or {}).items()
316
+ },
317
+ )
318
+ return built