smart-data-engine-sdk 0.1.0.dev0__py3-none-any.whl
This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
- sde/__init__.py +226 -0
- sde/canonical.py +141 -0
- sde/capabilities.py +62 -0
- sde/engines/__init__.py +0 -0
- sde/engines/clickhouse.py +689 -0
- sde/engines/orderbook.py +454 -0
- sde/engines/postgres.py +672 -0
- sde/entity.py +170 -0
- sde/errors.py +88 -0
- sde/explain.py +300 -0
- sde/groups.py +97 -0
- sde/hashing.py +242 -0
- sde/infer.py +461 -0
- sde/internal.py +90 -0
- sde/layout.py +660 -0
- sde/logging.py +132 -0
- sde/migration.py +820 -0
- sde/model.py +482 -0
- sde/placement.py +818 -0
- sde/py.typed +0 -0
- sde/routing.py +85 -0
- sde/schema.py +370 -0
- sde/session.py +507 -0
- sde/shapes.py +153 -0
- sde/telemetry.py +736 -0
- sde/testing/__init__.py +14 -0
- sde/testing/loader.py +175 -0
- sde/testing/memory.py +318 -0
- sde/types.py +228 -0
- sde/watermark.py +222 -0
- smart_data_engine_sdk-0.1.0.dev0.dist-info/METADATA +152 -0
- smart_data_engine_sdk-0.1.0.dev0.dist-info/RECORD +35 -0
- smart_data_engine_sdk-0.1.0.dev0.dist-info/WHEEL +4 -0
- smart_data_engine_sdk-0.1.0.dev0.dist-info/licenses/LICENSE +201 -0
- smart_data_engine_sdk-0.1.0.dev0.dist-info/licenses/NOTICE +13 -0
sde/testing/__init__.py
ADDED
|
@@ -0,0 +1,14 @@
|
|
|
1
|
+
"""Support for running the shared conformance vectors, and for testing an adapter.
|
|
2
|
+
|
|
3
|
+
Two modules, both public and both narrow. :mod:`sde.testing.loader` turns a vector's neutral
|
|
4
|
+
declaration into a model, which every implementation needs because a vector cannot contain one
|
|
5
|
+
language's classes. :mod:`sde.testing.memory` is an in-memory engine, which every implementation
|
|
6
|
+
needs because the ``migration/`` vectors pin behaviour that only happens against one.
|
|
7
|
+
"""
|
|
8
|
+
|
|
9
|
+
from __future__ import annotations
|
|
10
|
+
|
|
11
|
+
from .loader import model_from_neutral
|
|
12
|
+
from .memory import MemoryEngine, Recorded, engines_from
|
|
13
|
+
|
|
14
|
+
__all__ = ["MemoryEngine", "Recorded", "engines_from", "model_from_neutral"]
|
sde/testing/loader.py
ADDED
|
@@ -0,0 +1,175 @@
|
|
|
1
|
+
"""Building a model from neutral JSON, so that conformance vectors can be shared.
|
|
2
|
+
|
|
3
|
+
Every library needs this. A vector cannot contain Python decorators or TypeScript classes, so the
|
|
4
|
+
declaration in a vector is plain JSON and each implementation needs a way to turn that JSON into
|
|
5
|
+
whatever its own model type is. It is a small amount of code and it is a requirement, not a
|
|
6
|
+
convenience: without it the vectors could not be common, and without common vectors four
|
|
7
|
+
implementations drift apart silently.
|
|
8
|
+
|
|
9
|
+
It deliberately does *not* re-implement the encoding. It produces the same specs the decorator path
|
|
10
|
+
produces and hands them to :func:`sde.model.assemble`. A loader that built its own IR would make the
|
|
11
|
+
vectors verify a code path no application ever executes, which is the most expensive kind of green
|
|
12
|
+
test - it looks like coverage and is the absence of it.
|
|
13
|
+
"""
|
|
14
|
+
|
|
15
|
+
from __future__ import annotations
|
|
16
|
+
|
|
17
|
+
from collections.abc import Mapping
|
|
18
|
+
from typing import Any
|
|
19
|
+
|
|
20
|
+
from ..errors import DeclarationError
|
|
21
|
+
from ..model import EntitySpec, FieldSpec, LogicalModel, RelationSpec, assemble
|
|
22
|
+
from ..types import NEUTRAL_TYPES
|
|
23
|
+
|
|
24
|
+
__all__ = ["model_from_neutral"]
|
|
25
|
+
|
|
26
|
+
_DECIMAL_PREFIX = "decimal("
|
|
27
|
+
|
|
28
|
+
|
|
29
|
+
def _check_type(neutral: str, where: str) -> str:
|
|
30
|
+
if neutral in NEUTRAL_TYPES:
|
|
31
|
+
return neutral
|
|
32
|
+
|
|
33
|
+
# Anything beginning with "decimal" gets the decimal-specific message, including a bare
|
|
34
|
+
# "decimal" with no parameters at all. A conformance vector caught this: the generic "not in the
|
|
35
|
+
# vocabulary" message is true and useless, because the reader's actual mistake is that they left
|
|
36
|
+
# off the precision, and telling them the word is unknown sends them looking for the right word.
|
|
37
|
+
if neutral == "decimal" or neutral.startswith("decimal"):
|
|
38
|
+
if neutral.startswith(_DECIMAL_PREFIX) and neutral.endswith(")"):
|
|
39
|
+
body = neutral[len(_DECIMAL_PREFIX) : -1]
|
|
40
|
+
parts = body.split(",")
|
|
41
|
+
if len(parts) == 2 and all(p.strip().isdigit() for p in parts):
|
|
42
|
+
digits, scale = (int(p) for p in parts)
|
|
43
|
+
if body == f"{digits},{scale}" and digits >= 1 and 0 <= scale <= digits:
|
|
44
|
+
return neutral
|
|
45
|
+
raise DeclarationError(
|
|
46
|
+
f"{where}: {neutral!r} is not a well-formed decimal. The written form is "
|
|
47
|
+
"decimal(digits,scale) - precision then scale, no spaces, both required. No spaces "
|
|
48
|
+
"because whitespace inside a type name is exactly the sort of thing two libraries "
|
|
49
|
+
"would "
|
|
50
|
+
"disagree about, and both required because a decimal without precision is not a "
|
|
51
|
+
"storable type in any engine we place data in."
|
|
52
|
+
)
|
|
53
|
+
|
|
54
|
+
raise DeclarationError(
|
|
55
|
+
f"{where}: {neutral!r} is not in the neutral type vocabulary "
|
|
56
|
+
f"({', '.join(sorted(NEUTRAL_TYPES))}, decimal(p,s))"
|
|
57
|
+
)
|
|
58
|
+
|
|
59
|
+
|
|
60
|
+
def _check_shape(raw: Mapping[str, Any], name: str) -> None:
|
|
61
|
+
"""The shapes a person can plausibly hand this function, named rather than crashed on.
|
|
62
|
+
|
|
63
|
+
The first one is the whole reason this exists. The IR and the neutral form are close enough to
|
|
64
|
+
be confused - our own tests declared a model's IR to the control plane and got away with it for
|
|
65
|
+
weeks, because nothing read the stored document back - and they differ exactly where a key is
|
|
66
|
+
written. Handing the IR over used to reach a set membership test on a dict and raise
|
|
67
|
+
``TypeError: unhashable type: 'dict'`` from inside the encoder, which tells the reader nothing
|
|
68
|
+
about the mistake they made. Vector ``errors/036``.
|
|
69
|
+
"""
|
|
70
|
+
fields = raw.get("fields")
|
|
71
|
+
if not isinstance(fields, list):
|
|
72
|
+
raise DeclarationError(f"{name}: 'fields' is a list, and this one is {fields!r}")
|
|
73
|
+
for field in fields:
|
|
74
|
+
if not isinstance(field, Mapping) or not isinstance(field.get("name"), str):
|
|
75
|
+
raise DeclarationError(f"{name}: a field is an object with a name, not {field!r}")
|
|
76
|
+
if not isinstance(field.get("type"), str):
|
|
77
|
+
raise DeclarationError(f"{name}.{field['name']}: a field needs a type")
|
|
78
|
+
|
|
79
|
+
key = raw.get("key")
|
|
80
|
+
if key is not None and not isinstance(key, list):
|
|
81
|
+
raise DeclarationError(f"{name}: 'key' is a list of field names, not {key!r}")
|
|
82
|
+
for part in key or ():
|
|
83
|
+
if isinstance(part, Mapping) and "field" in part and "position" in part:
|
|
84
|
+
raise DeclarationError(
|
|
85
|
+
f"{name}: 'key' holds {part!r}, which is the IR's key form rather than the neutral "
|
|
86
|
+
"one. The IR records a position because array order is not load-bearing anywhere "
|
|
87
|
+
"else in it; a declaration states a key as a list of field names, in order. If you "
|
|
88
|
+
"meant to hand over a model you already built, sde.neutral_declaration() produces "
|
|
89
|
+
"this document from it."
|
|
90
|
+
)
|
|
91
|
+
if not isinstance(part, str):
|
|
92
|
+
raise DeclarationError(f"{name}: 'key' names fields as strings, not {part!r}")
|
|
93
|
+
|
|
94
|
+
for plural in ("pii",):
|
|
95
|
+
value = raw.get(plural)
|
|
96
|
+
if value is not None and (
|
|
97
|
+
not isinstance(value, list) or any(not isinstance(v, str) for v in value)
|
|
98
|
+
):
|
|
99
|
+
raise DeclarationError(f"{name}: {plural!r} is a list of field names, not {value!r}")
|
|
100
|
+
|
|
101
|
+
|
|
102
|
+
def model_from_neutral(data: Mapping[str, Any]) -> LogicalModel:
|
|
103
|
+
"""Build a :class:`~sde.model.LogicalModel` from a vector's ``model.json``.
|
|
104
|
+
|
|
105
|
+
The neutral form states keys as a plain list, because that is what a human writes. Turning it
|
|
106
|
+
into the positioned form the IR uses is this library's job, which is the point: if the vector
|
|
107
|
+
carried the positioned form we would be checking that we can copy JSON.
|
|
108
|
+
|
|
109
|
+
The shape of the document is checked before anything is read out of it. Not defensiveness:
|
|
110
|
+
this is the entry point a client's own tooling hits and the one a new implementation writes
|
|
111
|
+
first, and a bare ``KeyError`` out of a dictionary lookup is the failure mode section 7 of the
|
|
112
|
+
format contract already names for a missing rule - a library exception with no explanation, on
|
|
113
|
+
the path whose whole job is to explain.
|
|
114
|
+
"""
|
|
115
|
+
if not isinstance(data, Mapping):
|
|
116
|
+
raise DeclarationError("a neutral model declaration is an object with an 'entities' list")
|
|
117
|
+
raw_entities = data.get("entities")
|
|
118
|
+
if not isinstance(raw_entities, list):
|
|
119
|
+
raise DeclarationError("a neutral model declaration needs an 'entities' list")
|
|
120
|
+
|
|
121
|
+
entities: list[EntitySpec] = []
|
|
122
|
+
for raw in raw_entities:
|
|
123
|
+
if not isinstance(raw, Mapping):
|
|
124
|
+
raise DeclarationError(f"an entity is an object, and this one is {raw!r}")
|
|
125
|
+
if not isinstance(raw.get("name"), str) or not raw["name"]:
|
|
126
|
+
raise DeclarationError(f"an entity needs a name, and this one has {raw.get('name')!r}")
|
|
127
|
+
name = raw["name"]
|
|
128
|
+
_check_shape(raw, name)
|
|
129
|
+
fields = tuple(
|
|
130
|
+
FieldSpec(
|
|
131
|
+
name=f["name"],
|
|
132
|
+
type=_check_type(f["type"], f"{name}.{f['name']}"),
|
|
133
|
+
nullable=bool(f.get("nullable", False)),
|
|
134
|
+
)
|
|
135
|
+
for f in raw.get("fields", ())
|
|
136
|
+
)
|
|
137
|
+
# No default for the key. This line used to read `raw.get("key") or ("id",)` and the
|
|
138
|
+
# TypeScript port spelled the same thing with `??`, which differs on exactly one input:
|
|
139
|
+
# `"key": []` invented a key here and stayed keyless there, so one declaration had two
|
|
140
|
+
# model versions and no vector could see it. The rule is format-contract §4a now, and every
|
|
141
|
+
# refusal about a declaration lives in `assemble`, where both front doors meet - this one
|
|
142
|
+
# enforced a subset and the vectors therefore ran a weaker validator than any application.
|
|
143
|
+
key = tuple(raw.get("key") or ())
|
|
144
|
+
entities.append(
|
|
145
|
+
EntitySpec(
|
|
146
|
+
name=name,
|
|
147
|
+
fields=tuple(sorted(fields, key=lambda f: f.name)),
|
|
148
|
+
key=key,
|
|
149
|
+
pii=tuple(sorted(raw.get("pii") or ())),
|
|
150
|
+
residency=raw.get("residency"),
|
|
151
|
+
)
|
|
152
|
+
)
|
|
153
|
+
|
|
154
|
+
names = {e.name for e in entities}
|
|
155
|
+
relations: list[RelationSpec] = []
|
|
156
|
+
for raw in data.get("relations", ()):
|
|
157
|
+
source, target = raw["from"], raw["to"]
|
|
158
|
+
for side in (source, target):
|
|
159
|
+
if side not in names:
|
|
160
|
+
raise DeclarationError(f"relation {raw['name']!r} names unknown entity {side!r}")
|
|
161
|
+
relations.append(RelationSpec(name=raw["name"], source=source, target=target))
|
|
162
|
+
|
|
163
|
+
atomic_raw = data.get("atomic") or ()
|
|
164
|
+
atomic = tuple(sorted(tuple(sorted(group)) for group in atomic_raw))
|
|
165
|
+
for group in atomic:
|
|
166
|
+
unknown = [m for m in group if m not in names]
|
|
167
|
+
if unknown:
|
|
168
|
+
raise DeclarationError(f"atomic group names unknown entities {unknown}")
|
|
169
|
+
|
|
170
|
+
return assemble(
|
|
171
|
+
entities=tuple(entities),
|
|
172
|
+
relations=tuple(relations),
|
|
173
|
+
atomic=atomic,
|
|
174
|
+
cost_ceiling=data.get("cost_ceiling"),
|
|
175
|
+
)
|
sde/testing/memory.py
ADDED
|
@@ -0,0 +1,318 @@
|
|
|
1
|
+
"""An in-memory engine, for the ``migration/`` conformance vectors and for anybody's adapter tests.
|
|
2
|
+
|
|
3
|
+
**Why this is in the library rather than in a test file.** The ``migration/`` vectors pin behaviour
|
|
4
|
+
that only happens against an engine: the order of the calls a backfill makes, the arithmetic of the
|
|
5
|
+
resume marker, which side of the marker a verify counter lands on. Every implementation therefore
|
|
6
|
+
needs an engine to run them against, and the two options were for each runner to write its own or
|
|
7
|
+
for the fixture to be shared. A runner that writes its own is a runner whose *fixture* can be the
|
|
8
|
+
thing that differs, and then a red vector means "one of two tables disagreed" rather than "one of
|
|
9
|
+
two libraries disagreed" - which is the failure the whole suite exists to avoid, one level down.
|
|
10
|
+
|
|
11
|
+
**What it is not.** It is not an implementation of anything in the format contract. It stores rows
|
|
12
|
+
in a list and answers questions about them; every rule the vectors check lives in
|
|
13
|
+
:mod:`sde.migration` and in :mod:`sde.watermark`. The one property it does have to get right is a
|
|
14
|
+
keyset scan, and the vectors pin the *calls* as well as the results, so a fixture that scanned
|
|
15
|
+
differently would show up as a different call sequence rather than as a plausible wrong answer.
|
|
16
|
+
|
|
17
|
+
It is also genuinely useful outside the vectors, which is why it is public: an adapter written
|
|
18
|
+
against :class:`sde.migration.Migratable` has the same behaviour to check, and checking it against a
|
|
19
|
+
real server is slow and checking it against nothing is what leaves a backfill copying the same chunk
|
|
20
|
+
forever.
|
|
21
|
+
"""
|
|
22
|
+
|
|
23
|
+
from __future__ import annotations
|
|
24
|
+
|
|
25
|
+
from collections.abc import Iterator, Mapping, Sequence
|
|
26
|
+
from contextlib import contextmanager
|
|
27
|
+
from typing import Any
|
|
28
|
+
|
|
29
|
+
from ..errors import EngineError
|
|
30
|
+
from ..placement import PhysicalLayout
|
|
31
|
+
|
|
32
|
+
__all__ = ["MemoryEngine", "Recorded", "engines_from"]
|
|
33
|
+
|
|
34
|
+
|
|
35
|
+
class Recorded:
|
|
36
|
+
"""The calls a set of engines received, in one sequence, each entry naming its engine.
|
|
37
|
+
|
|
38
|
+
Kept because the ``migration/`` vectors pin the sequence and not only the outcome. A library
|
|
39
|
+
that reached the same progress record by scanning the whole table and filtering in memory would
|
|
40
|
+
satisfy every count and be unusable on a real table; the calls are the part that says how the
|
|
41
|
+
result was obtained.
|
|
42
|
+
|
|
43
|
+
**One journal for the whole engine set, and that is a fix.** The first version kept a list per
|
|
44
|
+
engine, which cannot express the guarantee the dual-write cases are about: a row reaches the
|
|
45
|
+
source *before* anything is attempted against the copy. Reversing those two lines passed every
|
|
46
|
+
vector, because each engine's own list was still in order - and the vector's own note claimed
|
|
47
|
+
that ordering was what it pinned. A guarantee across two engines needs one sequence.
|
|
48
|
+
"""
|
|
49
|
+
|
|
50
|
+
def __init__(self) -> None:
|
|
51
|
+
self.calls: list[dict[str, Any]] = []
|
|
52
|
+
|
|
53
|
+
def note(self, engine: str, method: str, **arguments: Any) -> None:
|
|
54
|
+
self.calls.append({"engine": engine, "call": method, **arguments})
|
|
55
|
+
|
|
56
|
+
def as_list(self) -> list[dict[str, Any]]:
|
|
57
|
+
return [dict(entry) for entry in self.calls]
|
|
58
|
+
|
|
59
|
+
|
|
60
|
+
def _sortable(value: Any) -> tuple[int, Any]:
|
|
61
|
+
"""A total order over the value kinds a vector may use, with the kind first.
|
|
62
|
+
|
|
63
|
+
Comparing a string to an integer raises in Python and coerces in JavaScript, and neither is a
|
|
64
|
+
key order. Vectors use one kind per column, so this never has to decide *between* kinds for a
|
|
65
|
+
real comparison - the tag is there so that a vector which accidentally mixed them fails loudly
|
|
66
|
+
in both languages instead of one.
|
|
67
|
+
"""
|
|
68
|
+
if value is None:
|
|
69
|
+
return (0, 0)
|
|
70
|
+
if isinstance(value, bool):
|
|
71
|
+
return (1, int(value))
|
|
72
|
+
if isinstance(value, (int, float)):
|
|
73
|
+
return (2, value)
|
|
74
|
+
return (3, str(value))
|
|
75
|
+
|
|
76
|
+
|
|
77
|
+
def _key(order: Sequence[str], row: Mapping[str, Any]) -> tuple[tuple[int, Any], ...]:
|
|
78
|
+
return tuple(_sortable(row.get(column)) for column in order)
|
|
79
|
+
|
|
80
|
+
|
|
81
|
+
class MemoryEngine:
|
|
82
|
+
"""One dialect, a set of named tables, and the two optional protocols an adapter may offer.
|
|
83
|
+
|
|
84
|
+
``dialect`` matters: the migration gate refuses a copy between two dialects whose precision
|
|
85
|
+
differs, so a fixture that reported one dialect for both ends could not reach that refusal.
|
|
86
|
+
|
|
87
|
+
``can_keep_bookkeeping`` and ``can_migrate`` exist so a vector can build an engine that
|
|
88
|
+
**cannot** take part, which is the case both of those refusals are about. They remove the
|
|
89
|
+
methods rather than making them fail, because that is what a real adapter without the capability
|
|
90
|
+
looks like - and the check is "will this object answer these calls".
|
|
91
|
+
"""
|
|
92
|
+
|
|
93
|
+
def __init__(
|
|
94
|
+
self,
|
|
95
|
+
dialect: str = "postgres",
|
|
96
|
+
*,
|
|
97
|
+
name: str = "engine",
|
|
98
|
+
journal: Recorded | None = None,
|
|
99
|
+
tables: Mapping[str, Sequence[Mapping[str, Any]]] | None = None,
|
|
100
|
+
can_keep_bookkeeping: bool = True,
|
|
101
|
+
can_migrate: bool = True,
|
|
102
|
+
watermark: int | None = None,
|
|
103
|
+
markers: Mapping[tuple[str, str], int] | None = None,
|
|
104
|
+
fail_inserts: Mapping[str, int] | None = None,
|
|
105
|
+
) -> None:
|
|
106
|
+
self.dialect = dialect
|
|
107
|
+
self.tables: dict[str, list[dict[str, Any]]] = {
|
|
108
|
+
name: [dict(row) for row in rows] for name, rows in (tables or {}).items()
|
|
109
|
+
}
|
|
110
|
+
self.name = name
|
|
111
|
+
self.recorded = journal if journal is not None else Recorded()
|
|
112
|
+
# How many of the next inserts into each table must fail. The one thing a fake has to be
|
|
113
|
+
# able to do that a real engine does on its own: a fan-out that does not reach the copy is
|
|
114
|
+
# the case the whole dual-write design is about, and it cannot be reached by writing
|
|
115
|
+
# correct rows to a working table.
|
|
116
|
+
self._fail_inserts: dict[str, int] = dict(fail_inserts or {})
|
|
117
|
+
self._watermarks: list[int] = [] if watermark is None else [watermark]
|
|
118
|
+
self._markers: dict[tuple[str, str], list[int]] = {
|
|
119
|
+
key: [value] for key, value in (markers or {}).items()
|
|
120
|
+
}
|
|
121
|
+
# **Bound onto the instance when the capability is on, and genuinely absent when it is
|
|
122
|
+
# off.** Setting them to ``None`` instead was the first attempt and it does not work:
|
|
123
|
+
# ``satisfies`` asks ``hasattr``, deliberately - a member that exists and is not callable
|
|
124
|
+
# fails at the call with a message naming it, which is a better failure than a capability
|
|
125
|
+
# check that quietly answers "no". So an attribute set to None reads as *present*, and the
|
|
126
|
+
# engine that was supposed to be unable to take part took part and crashed. Absence is also
|
|
127
|
+
# what a real adapter without the capability looks like.
|
|
128
|
+
if can_keep_bookkeeping:
|
|
129
|
+
self.map_watermark = self._map_watermark
|
|
130
|
+
self.record_map_version = self._record_map_version
|
|
131
|
+
if can_migrate:
|
|
132
|
+
self.key_range = self._key_range
|
|
133
|
+
self.nth_key = self._nth_key
|
|
134
|
+
self.copy_in = self._copy_in
|
|
135
|
+
self.count = self._count
|
|
136
|
+
self.backfill_marker = self._backfill_marker
|
|
137
|
+
self.record_backfill_marker = self._record_backfill_marker
|
|
138
|
+
|
|
139
|
+
# --- Engine ------------------------------------------------------------------------------
|
|
140
|
+
|
|
141
|
+
def ensure_schema(self, layout: PhysicalLayout, *, keys: Mapping[str, Any]) -> None:
|
|
142
|
+
self.recorded.note(self.name, "ensure_schema", tables=sorted(layout.tables.values()))
|
|
143
|
+
for table in layout.tables.values():
|
|
144
|
+
self.tables.setdefault(table, [])
|
|
145
|
+
|
|
146
|
+
def insert(self, table: str, values: Mapping[str, Any]) -> None:
|
|
147
|
+
self.recorded.note(self.name, "insert", table=table)
|
|
148
|
+
remaining = self._fail_inserts.get(table, 0)
|
|
149
|
+
if remaining > 0:
|
|
150
|
+
self._fail_inserts[table] = remaining - 1
|
|
151
|
+
raise EngineError(f"insert into {table} failed: this engine was told to refuse it")
|
|
152
|
+
self.tables.setdefault(table, []).append(dict(values))
|
|
153
|
+
|
|
154
|
+
def get(self, table: str, key: Mapping[str, Any]) -> dict[str, Any] | None:
|
|
155
|
+
self.recorded.note(self.name, "get", table=table)
|
|
156
|
+
for row in self.tables.get(table, []):
|
|
157
|
+
if all(row.get(column) == value for column, value in key.items()):
|
|
158
|
+
return dict(row)
|
|
159
|
+
return None
|
|
160
|
+
|
|
161
|
+
@contextmanager
|
|
162
|
+
def transaction(self) -> Iterator[MemoryEngine]:
|
|
163
|
+
"""Snapshot, yield, and put the snapshot back on failure.
|
|
164
|
+
|
|
165
|
+
Enough to make a rollback observable, which is what the dual-write tests need: rows written
|
|
166
|
+
inside a transaction that raises must not reach the copy, and the only way to check that is
|
|
167
|
+
for the source to forget them too.
|
|
168
|
+
"""
|
|
169
|
+
self.recorded.note(self.name, "transaction")
|
|
170
|
+
snapshot = {name: [dict(row) for row in rows] for name, rows in self.tables.items()}
|
|
171
|
+
try:
|
|
172
|
+
yield self
|
|
173
|
+
except BaseException:
|
|
174
|
+
self.tables = snapshot
|
|
175
|
+
raise
|
|
176
|
+
|
|
177
|
+
# --- WatermarkStore ----------------------------------------------------------------------
|
|
178
|
+
|
|
179
|
+
def _map_watermark(self) -> int | None:
|
|
180
|
+
self.recorded.note(self.name, "map_watermark")
|
|
181
|
+
return max(self._watermarks) if self._watermarks else None
|
|
182
|
+
|
|
183
|
+
def _record_map_version(self, version: int, *, model_version: str) -> None:
|
|
184
|
+
self.recorded.note(self.name, "record_map_version", version=version)
|
|
185
|
+
self._watermarks.append(version)
|
|
186
|
+
|
|
187
|
+
# --- Migratable --------------------------------------------------------------------------
|
|
188
|
+
|
|
189
|
+
def _key_range(
|
|
190
|
+
self,
|
|
191
|
+
table: str,
|
|
192
|
+
order: Sequence[str],
|
|
193
|
+
*,
|
|
194
|
+
after: Sequence[Any] | None = None,
|
|
195
|
+
upto: Sequence[Any] | None = None,
|
|
196
|
+
limit: int | None = None,
|
|
197
|
+
) -> list[dict[str, Any]]:
|
|
198
|
+
from ..migration import key_columns, same_width
|
|
199
|
+
|
|
200
|
+
cols = key_columns(order, table)
|
|
201
|
+
if after is not None:
|
|
202
|
+
same_width(after, cols, "after")
|
|
203
|
+
if upto is not None:
|
|
204
|
+
same_width(upto, cols, "upto")
|
|
205
|
+
self.recorded.note(
|
|
206
|
+
self.name,
|
|
207
|
+
"key_range",
|
|
208
|
+
table=table,
|
|
209
|
+
after=None if after is None else list(after),
|
|
210
|
+
upto=None if upto is None else list(upto),
|
|
211
|
+
limit=limit,
|
|
212
|
+
)
|
|
213
|
+
rows = sorted(self.tables.get(table, []), key=lambda row: _key(cols, row))
|
|
214
|
+
if after is not None:
|
|
215
|
+
low = tuple(_sortable(value) for value in after)
|
|
216
|
+
rows = [row for row in rows if _key(cols, row) > low]
|
|
217
|
+
if upto is not None:
|
|
218
|
+
high = tuple(_sortable(value) for value in upto)
|
|
219
|
+
rows = [row for row in rows if _key(cols, row) <= high]
|
|
220
|
+
if limit is not None:
|
|
221
|
+
rows = rows[:limit]
|
|
222
|
+
return [dict(row) for row in rows]
|
|
223
|
+
|
|
224
|
+
def _nth_key(
|
|
225
|
+
self, table: str, order: Sequence[str], *, position: int
|
|
226
|
+
) -> tuple[Any, ...] | None:
|
|
227
|
+
from ..migration import key_columns
|
|
228
|
+
|
|
229
|
+
cols = key_columns(order, table)
|
|
230
|
+
self.recorded.note(self.name, "nth_key", table=table, position=position)
|
|
231
|
+
if position < 1:
|
|
232
|
+
raise EngineError(f"position is one-based; {position} is not a row")
|
|
233
|
+
rows = sorted(self.tables.get(table, []), key=lambda row: _key(cols, row))
|
|
234
|
+
if position > len(rows):
|
|
235
|
+
return None
|
|
236
|
+
row = rows[position - 1]
|
|
237
|
+
return tuple(row.get(column) for column in cols)
|
|
238
|
+
|
|
239
|
+
def _copy_in(self, table: str, rows: Sequence[Mapping[str, Any]]) -> None:
|
|
240
|
+
self.recorded.note(self.name, "copy_in", table=table, rows=len(rows))
|
|
241
|
+
if not rows:
|
|
242
|
+
return
|
|
243
|
+
columns = sorted(rows[0])
|
|
244
|
+
for row in rows:
|
|
245
|
+
if sorted(row) != columns:
|
|
246
|
+
raise EngineError(
|
|
247
|
+
f"copy_in into {table} was given rows with different columns "
|
|
248
|
+
f"({columns} and {sorted(row)}). A chunk comes from one table, so this is a "
|
|
249
|
+
f"caller assembling it from two."
|
|
250
|
+
)
|
|
251
|
+
existing = self.tables.setdefault(table, [])
|
|
252
|
+
# Idempotent on the whole row's identity, which is what both real targets do by a different
|
|
253
|
+
# mechanism: ON CONFLICT DO NOTHING in PostgreSQL and a ReplacingMergeTree collapse in
|
|
254
|
+
# ClickHouse. Not the same mechanism, which is why the adapters have live tests in every
|
|
255
|
+
# direction and this only has to absorb a recopy.
|
|
256
|
+
for row in rows:
|
|
257
|
+
same = (
|
|
258
|
+
all(present.get(column) == row.get(column) for column in row)
|
|
259
|
+
for present in existing
|
|
260
|
+
)
|
|
261
|
+
if not any(same):
|
|
262
|
+
existing.append(dict(row))
|
|
263
|
+
|
|
264
|
+
def _count(self, table: str) -> int:
|
|
265
|
+
self.recorded.note(self.name, "count", table=table)
|
|
266
|
+
return len(self.tables.get(table, []))
|
|
267
|
+
|
|
268
|
+
def _backfill_marker(self, *, materialization: str, entity: str) -> int:
|
|
269
|
+
self.recorded.note(
|
|
270
|
+
self.name, "backfill_marker", materialization=materialization, entity=entity
|
|
271
|
+
)
|
|
272
|
+
seen = self._markers.get((materialization, entity))
|
|
273
|
+
return max(seen) if seen else 0
|
|
274
|
+
|
|
275
|
+
def _record_backfill_marker(self, *, materialization: str, entity: str, rows: int) -> None:
|
|
276
|
+
self.recorded.note(
|
|
277
|
+
self.name,
|
|
278
|
+
"record_backfill_marker",
|
|
279
|
+
materialization=materialization,
|
|
280
|
+
entity=entity,
|
|
281
|
+
rows=rows,
|
|
282
|
+
)
|
|
283
|
+
self._markers.setdefault((materialization, entity), []).append(rows)
|
|
284
|
+
|
|
285
|
+
|
|
286
|
+
def engines_from(
|
|
287
|
+
spec: Mapping[str, Mapping[str, Any]], journal: Recorded | None = None
|
|
288
|
+
) -> dict[str, MemoryEngine]:
|
|
289
|
+
"""Build the engine set a ``migration/`` case describes.
|
|
290
|
+
|
|
291
|
+
Here rather than in each runner for the same reason the engine itself is here: two runners that
|
|
292
|
+
each read this document their own way can disagree about the *fixture*, and then a red vector
|
|
293
|
+
says "one of two tables differed" instead of "one of two libraries differed". The generator uses
|
|
294
|
+
this too, so the document a vector carries is read by exactly one piece of code per language.
|
|
295
|
+
"""
|
|
296
|
+
shared = journal if journal is not None else Recorded()
|
|
297
|
+
built: dict[str, MemoryEngine] = {}
|
|
298
|
+
for name, body in spec.items():
|
|
299
|
+
built[name] = MemoryEngine(
|
|
300
|
+
dialect=str(body.get("dialect", "postgres")),
|
|
301
|
+
name=name,
|
|
302
|
+
journal=shared,
|
|
303
|
+
tables={
|
|
304
|
+
table: [dict(row) for row in rows]
|
|
305
|
+
for table, rows in (body.get("tables") or {}).items()
|
|
306
|
+
},
|
|
307
|
+
can_keep_bookkeeping=bool(body.get("bookkeeping", True)),
|
|
308
|
+
can_migrate=bool(body.get("migratable", True)),
|
|
309
|
+
watermark=body.get("watermark"),
|
|
310
|
+
markers={
|
|
311
|
+
(key.split("|", 1)[0], key.split("|", 1)[1]): int(value)
|
|
312
|
+
for key, value in (body.get("markers") or {}).items()
|
|
313
|
+
},
|
|
314
|
+
fail_inserts={
|
|
315
|
+
table: int(count) for table, count in (body.get("fail_inserts") or {}).items()
|
|
316
|
+
},
|
|
317
|
+
)
|
|
318
|
+
return built
|