smart-data-engine-sdk 0.1.0.dev0__py3-none-any.whl
This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
- sde/__init__.py +226 -0
- sde/canonical.py +141 -0
- sde/capabilities.py +62 -0
- sde/engines/__init__.py +0 -0
- sde/engines/clickhouse.py +689 -0
- sde/engines/orderbook.py +454 -0
- sde/engines/postgres.py +672 -0
- sde/entity.py +170 -0
- sde/errors.py +88 -0
- sde/explain.py +300 -0
- sde/groups.py +97 -0
- sde/hashing.py +242 -0
- sde/infer.py +461 -0
- sde/internal.py +90 -0
- sde/layout.py +660 -0
- sde/logging.py +132 -0
- sde/migration.py +820 -0
- sde/model.py +482 -0
- sde/placement.py +818 -0
- sde/py.typed +0 -0
- sde/routing.py +85 -0
- sde/schema.py +370 -0
- sde/session.py +507 -0
- sde/shapes.py +153 -0
- sde/telemetry.py +736 -0
- sde/testing/__init__.py +14 -0
- sde/testing/loader.py +175 -0
- sde/testing/memory.py +318 -0
- sde/types.py +228 -0
- sde/watermark.py +222 -0
- smart_data_engine_sdk-0.1.0.dev0.dist-info/METADATA +152 -0
- smart_data_engine_sdk-0.1.0.dev0.dist-info/RECORD +35 -0
- smart_data_engine_sdk-0.1.0.dev0.dist-info/WHEEL +4 -0
- smart_data_engine_sdk-0.1.0.dev0.dist-info/licenses/LICENSE +201 -0
- smart_data_engine_sdk-0.1.0.dev0.dist-info/licenses/NOTICE +13 -0
sde/infer.py
ADDED
|
@@ -0,0 +1,461 @@
|
|
|
1
|
+
"""A logical model suggested from sample rows, and what it refuses to guess.
|
|
2
|
+
|
|
3
|
+
Declaring a model by hand is the honest interface and it is a bad first step. Somebody with a few
|
|
4
|
+
thousand rows of weather readings wants to point at them and be told what the schema should be,
|
|
5
|
+
and telling them to write out entities, fields and neutral types first is asking for the work
|
|
6
|
+
before the value.
|
|
7
|
+
|
|
8
|
+
**The rows never leave the process.** This module runs in the client's library, on the client's
|
|
9
|
+
machine, and produces a declaration - names and types, no values. That is not an incidental
|
|
10
|
+
property of where the code happens to sit: the control plane's promise is that it never sees a
|
|
11
|
+
row, and the only way to make an on-ramp from real data compatible with that promise is to do the
|
|
12
|
+
inference on the side that already has the data. It also means the inference is public code
|
|
13
|
+
anybody can read, which is worth more than any assurance about it.
|
|
14
|
+
|
|
15
|
+
Everything here is a **suggestion with its evidence attached**, and four things are refused
|
|
16
|
+
outright.
|
|
17
|
+
|
|
18
|
+
**The four invariants cannot be inferred, at all, ever.** Atomicity, residency, personal data and
|
|
19
|
+
the cost ceiling are facts about a business and its jurisdiction, and there is nothing in a column
|
|
20
|
+
of numbers that implies them. This matters more than the rest of the module: a declaration that
|
|
21
|
+
looks complete and carries no invariants produces a placement that ignores the client's legal
|
|
22
|
+
constraints and looks perfectly healthy doing it. So the output states what is still missing, by
|
|
23
|
+
name, and ``what_you_must_still_declare`` is not decoration.
|
|
24
|
+
|
|
25
|
+
**Uniqueness across a sample is not a key.** Five rows with distinct values in a column is
|
|
26
|
+
evidence of nothing, and a key chosen that way is a key that starts rejecting writes in week
|
|
27
|
+
three. Candidates are reported with the sample size next to them and the choice stays with the
|
|
28
|
+
client. The one exception is the convention the library already has: a field named ``id`` is the
|
|
29
|
+
key unless ``Meta.key`` says otherwise, which is a documented rule rather than something guessed
|
|
30
|
+
from this data.
|
|
31
|
+
|
|
32
|
+
**Text is not coerced by looking at it.** A column of ISO-8601 strings is very probably a
|
|
33
|
+
timestamp, and inferring one would be right until the row that is not, at which point the client's
|
|
34
|
+
writes fail against a column type derived from data they no longer have. Candidate refinements are
|
|
35
|
+
reported instead.
|
|
36
|
+
|
|
37
|
+
**A column with two types has no type.** An integer in one row and a string in another is not a
|
|
38
|
+
narrowing problem, it is two columns that were given one name, and picking either produces silent
|
|
39
|
+
truncation later.
|
|
40
|
+
"""
|
|
41
|
+
|
|
42
|
+
from __future__ import annotations
|
|
43
|
+
|
|
44
|
+
import datetime as dt
|
|
45
|
+
import decimal
|
|
46
|
+
import re
|
|
47
|
+
import uuid
|
|
48
|
+
from collections.abc import Iterable, Mapping, Sequence
|
|
49
|
+
from dataclasses import dataclass
|
|
50
|
+
from typing import Any
|
|
51
|
+
|
|
52
|
+
from .errors import DeclarationError
|
|
53
|
+
|
|
54
|
+
_ISO_8601 = re.compile(
|
|
55
|
+
r"^\d{4}-\d{2}-\d{2}([T ]\d{2}:\d{2}(:\d{2}(\.\d+)?)?(Z|[+-]\d{2}:?\d{2})?)?$"
|
|
56
|
+
)
|
|
57
|
+
_UUID_TEXT = re.compile(
|
|
58
|
+
r"^[0-9a-fA-F]{8}-[0-9a-fA-F]{4}-[0-9a-fA-F]{4}-[0-9a-fA-F]{4}-[0-9a-fA-F]{12}$"
|
|
59
|
+
)
|
|
60
|
+
|
|
61
|
+
# What a Python value maps to, with nothing name-based and nothing content-based in it.
|
|
62
|
+
_FROM_TYPE: Mapping[type, str] = {
|
|
63
|
+
bool: "bool",
|
|
64
|
+
int: "int64",
|
|
65
|
+
float: "float64",
|
|
66
|
+
str: "string",
|
|
67
|
+
bytes: "bytes",
|
|
68
|
+
uuid.UUID: "uuid",
|
|
69
|
+
dt.date: "date",
|
|
70
|
+
}
|
|
71
|
+
|
|
72
|
+
|
|
73
|
+
@dataclass(frozen=True)
|
|
74
|
+
class Note:
|
|
75
|
+
"""One thing the client should look at, with the evidence that produced it.
|
|
76
|
+
|
|
77
|
+
``field`` is empty for notes about an entity as a whole. ``severity`` has two values and no
|
|
78
|
+
middle one: ``blocking`` means the declaration will not build until it is resolved, and
|
|
79
|
+
``review`` means it will build and may be wrong. A third level would be a way to file
|
|
80
|
+
something as neither.
|
|
81
|
+
"""
|
|
82
|
+
|
|
83
|
+
severity: str
|
|
84
|
+
entity: str
|
|
85
|
+
field: str
|
|
86
|
+
what: str
|
|
87
|
+
|
|
88
|
+
def __post_init__(self) -> None:
|
|
89
|
+
if self.severity not in ("blocking", "review"):
|
|
90
|
+
raise ValueError(f"severity is 'blocking' or 'review', not {self.severity!r}")
|
|
91
|
+
|
|
92
|
+
def __str__(self) -> str:
|
|
93
|
+
where = f"{self.entity}.{self.field}" if self.field else self.entity
|
|
94
|
+
return f"[{self.severity}] {where}: {self.what}"
|
|
95
|
+
|
|
96
|
+
|
|
97
|
+
@dataclass(frozen=True)
|
|
98
|
+
class InferredModel:
|
|
99
|
+
"""A declaration suggested from data, and everything that is still the client's to say.
|
|
100
|
+
|
|
101
|
+
``declaration`` is the neutral form - the same shape a model vector uses - so it can be
|
|
102
|
+
written to a file, read, corrected and committed. That is the point: the output of this module
|
|
103
|
+
is meant to be edited by a person, not consumed silently by the next call.
|
|
104
|
+
"""
|
|
105
|
+
|
|
106
|
+
declaration: dict[str, Any]
|
|
107
|
+
sample_size: int
|
|
108
|
+
candidate_keys: Mapping[str, tuple[str, ...]]
|
|
109
|
+
notes: tuple[Note, ...]
|
|
110
|
+
|
|
111
|
+
@property
|
|
112
|
+
def blocking(self) -> tuple[Note, ...]:
|
|
113
|
+
return tuple(n for n in self.notes if n.severity == "blocking")
|
|
114
|
+
|
|
115
|
+
def what_you_must_still_declare(self) -> tuple[str, ...]:
|
|
116
|
+
"""The four invariants, always, plus anything blocking.
|
|
117
|
+
|
|
118
|
+
The invariants are listed unconditionally and not "if they look relevant". There is no
|
|
119
|
+
signal in the data that says whether two entities must change together or whether a column
|
|
120
|
+
is personal data, so their absence from this list would have to mean "we checked" - and
|
|
121
|
+
nothing checked.
|
|
122
|
+
"""
|
|
123
|
+
out = [
|
|
124
|
+
"atomicity: which entities must change together in one transaction. Nothing in your "
|
|
125
|
+
"data implies this and it decides which engines may hold them.",
|
|
126
|
+
"residency: which entities must stay in which jurisdiction. This is a legal constraint "
|
|
127
|
+
"and it is a hard exclusion, not a preference.",
|
|
128
|
+
"personal data: which fields are personal data. It decides what may be copied into an "
|
|
129
|
+
"analytical materialisation.",
|
|
130
|
+
"cost ceiling: the monthly figure a placement may not exceed, and its currency.",
|
|
131
|
+
]
|
|
132
|
+
out.extend(str(note) for note in self.blocking)
|
|
133
|
+
return tuple(out)
|
|
134
|
+
|
|
135
|
+
def summary(self) -> str:
|
|
136
|
+
lines = [
|
|
137
|
+
f"Inferred from {self.sample_size} row(s). This is a suggestion, not a declaration.",
|
|
138
|
+
"",
|
|
139
|
+
]
|
|
140
|
+
for entity in self.declaration["entities"]:
|
|
141
|
+
fields = ", ".join(f"{f['name']}: {f['type']}" for f in entity["fields"])
|
|
142
|
+
lines.append(f" {entity['name']}({fields})")
|
|
143
|
+
candidates = self.candidate_keys.get(entity["name"], ())
|
|
144
|
+
if candidates and len(candidates) == len(entity["fields"]):
|
|
145
|
+
# Every column unique tells you nothing about which one is a key, and printing the
|
|
146
|
+
# full list dressed as a finding invites somebody to pick the first entry. Same
|
|
147
|
+
# shape as a placement score of 1.0000 from a registry with one engine in it.
|
|
148
|
+
lines.append(
|
|
149
|
+
f" every column is unique across {self.sample_size} row(s), which says "
|
|
150
|
+
f"nothing about which is a key - a sample this size makes uniqueness free. "
|
|
151
|
+
f"The key is yours to declare."
|
|
152
|
+
)
|
|
153
|
+
elif candidates:
|
|
154
|
+
lines.append(
|
|
155
|
+
f" unique across the sample: {list(candidates)} - {self.sample_size} row(s) "
|
|
156
|
+
f"is not evidence of uniqueness, so the key is yours to choose"
|
|
157
|
+
)
|
|
158
|
+
lines += ["", "Still yours to declare:"]
|
|
159
|
+
lines += [f" - {item}" for item in self.what_you_must_still_declare()]
|
|
160
|
+
if any(n.severity == "review" for n in self.notes):
|
|
161
|
+
lines += ["", "Worth a look:"]
|
|
162
|
+
lines += [f" - {n}" for n in self.notes if n.severity == "review"]
|
|
163
|
+
return "\n".join(lines)
|
|
164
|
+
|
|
165
|
+
|
|
166
|
+
def _decimal_type(values: Sequence[decimal.Decimal]) -> str:
|
|
167
|
+
"""A decimal type wide enough for what was seen, and one that says it is a guess.
|
|
168
|
+
|
|
169
|
+
Precision from a sample is a floor rather than a fact, so the width is padded: a column
|
|
170
|
+
holding 99.99 today holds 1000.00 next month, and a `numeric(4,2)` derived from four observed
|
|
171
|
+
digits fails on that write rather than rounding it. Padding is stated in a note, because a
|
|
172
|
+
client who knows the real range should narrow it.
|
|
173
|
+
"""
|
|
174
|
+
digits = 1
|
|
175
|
+
scale = 0
|
|
176
|
+
for value in values:
|
|
177
|
+
_, digits_tuple, exponent = value.as_tuple()
|
|
178
|
+
if not isinstance(exponent, int):
|
|
179
|
+
continue
|
|
180
|
+
scale = max(scale, -exponent if exponent < 0 else 0)
|
|
181
|
+
digits = max(digits, len(digits_tuple))
|
|
182
|
+
return f"decimal({min(digits + 4, 38)},{scale})"
|
|
183
|
+
|
|
184
|
+
|
|
185
|
+
def _neutral_type(entity: str, name: str, values: list[Any]) -> tuple[str, list[Note]]:
|
|
186
|
+
notes: list[Note] = []
|
|
187
|
+
present = [v for v in values if v is not None]
|
|
188
|
+
if not present:
|
|
189
|
+
return "", [
|
|
190
|
+
Note(
|
|
191
|
+
severity="blocking",
|
|
192
|
+
entity=entity,
|
|
193
|
+
field=name,
|
|
194
|
+
what=(
|
|
195
|
+
"every sampled row has this empty, so there is no evidence of what it holds. "
|
|
196
|
+
"Declare its type or drop the column."
|
|
197
|
+
),
|
|
198
|
+
)
|
|
199
|
+
]
|
|
200
|
+
|
|
201
|
+
kinds = {type(v) for v in present}
|
|
202
|
+
# datetime is a subclass of date, so it has to be resolved before the mapping is consulted.
|
|
203
|
+
if any(isinstance(v, dt.datetime) for v in present):
|
|
204
|
+
kinds = {dt.datetime if isinstance(v, dt.datetime) else type(v) for v in present}
|
|
205
|
+
|
|
206
|
+
if len(kinds) > 1:
|
|
207
|
+
# int and float together is the one mixture with an answer: a column of 1, 2, 2.5 is a
|
|
208
|
+
# float column somebody happened to write whole numbers into. Everything else is two
|
|
209
|
+
# columns with one name.
|
|
210
|
+
if kinds <= {int, float} and bool not in kinds:
|
|
211
|
+
notes.append(
|
|
212
|
+
Note(
|
|
213
|
+
severity="review",
|
|
214
|
+
entity=entity,
|
|
215
|
+
field=name,
|
|
216
|
+
what=(
|
|
217
|
+
"whole numbers and fractions in the same column, read as float64. If this "
|
|
218
|
+
"is money, declare it as a decimal: binary floating point cannot represent "
|
|
219
|
+
"0.01 and a total that is out by a cent is a total nobody trusts."
|
|
220
|
+
),
|
|
221
|
+
)
|
|
222
|
+
)
|
|
223
|
+
return "float64", notes
|
|
224
|
+
return "", [
|
|
225
|
+
Note(
|
|
226
|
+
severity="blocking",
|
|
227
|
+
entity=entity,
|
|
228
|
+
field=name,
|
|
229
|
+
what=(
|
|
230
|
+
f"two different types in the sample ({sorted(k.__name__ for k in kinds)}). "
|
|
231
|
+
f"This is not a narrowing problem: it is two columns that were given one name, "
|
|
232
|
+
f"and choosing either one truncates the other silently."
|
|
233
|
+
),
|
|
234
|
+
)
|
|
235
|
+
]
|
|
236
|
+
|
|
237
|
+
kind = next(iter(kinds))
|
|
238
|
+
|
|
239
|
+
if kind is decimal.Decimal:
|
|
240
|
+
inferred = _decimal_type([v for v in present if isinstance(v, decimal.Decimal)])
|
|
241
|
+
notes.append(
|
|
242
|
+
Note(
|
|
243
|
+
severity="review",
|
|
244
|
+
entity=entity,
|
|
245
|
+
field=name,
|
|
246
|
+
what=(
|
|
247
|
+
f"{inferred} - the precision is padded past what the sample shows, because a "
|
|
248
|
+
f"width taken from a sample is a floor and the first wider value would fail to "
|
|
249
|
+
f"write. Narrow it if you know the real range."
|
|
250
|
+
),
|
|
251
|
+
)
|
|
252
|
+
)
|
|
253
|
+
return inferred, notes
|
|
254
|
+
|
|
255
|
+
if kind is dt.datetime:
|
|
256
|
+
aware = [v for v in present if isinstance(v, dt.datetime) and v.tzinfo is not None]
|
|
257
|
+
if len(aware) == len(present):
|
|
258
|
+
return "timestamptz", notes
|
|
259
|
+
if not aware:
|
|
260
|
+
notes.append(
|
|
261
|
+
Note(
|
|
262
|
+
severity="review",
|
|
263
|
+
entity=entity,
|
|
264
|
+
field=name,
|
|
265
|
+
what=(
|
|
266
|
+
"naive datetimes, read as timestamp without a zone. If these are instants "
|
|
267
|
+
"rather than wall-clock readings, declare timestamptz: a naive value means "
|
|
268
|
+
"different moments in two engines, and the difference is silent."
|
|
269
|
+
),
|
|
270
|
+
)
|
|
271
|
+
)
|
|
272
|
+
return "timestamp", notes
|
|
273
|
+
return "", [
|
|
274
|
+
Note(
|
|
275
|
+
severity="blocking",
|
|
276
|
+
entity=entity,
|
|
277
|
+
field=name,
|
|
278
|
+
what=(
|
|
279
|
+
"some rows carry a timezone and some do not. One column cannot be both, and "
|
|
280
|
+
"guessing turns a subset of your data into the wrong instant."
|
|
281
|
+
),
|
|
282
|
+
)
|
|
283
|
+
]
|
|
284
|
+
|
|
285
|
+
if kind is float:
|
|
286
|
+
notes.append(
|
|
287
|
+
Note(
|
|
288
|
+
severity="review",
|
|
289
|
+
entity=entity,
|
|
290
|
+
field=name,
|
|
291
|
+
what=(
|
|
292
|
+
"float64. If this is money or anything summed and compared for equality, "
|
|
293
|
+
"declare a decimal instead: binary floating point cannot represent 0.01."
|
|
294
|
+
),
|
|
295
|
+
)
|
|
296
|
+
)
|
|
297
|
+
return "float64", notes
|
|
298
|
+
|
|
299
|
+
if kind is str:
|
|
300
|
+
texts = [v for v in present if isinstance(v, str)]
|
|
301
|
+
if texts and all(_UUID_TEXT.match(v) for v in texts):
|
|
302
|
+
notes.append(
|
|
303
|
+
Note(
|
|
304
|
+
severity="review",
|
|
305
|
+
entity=entity,
|
|
306
|
+
field=name,
|
|
307
|
+
what=(
|
|
308
|
+
"every sampled value looks like a UUID, and is still read as a "
|
|
309
|
+
"string. Declare uuid if that is what it is - not inferred, because the "
|
|
310
|
+
"row that is not a UUID would fail to write against a type derived from "
|
|
311
|
+
"rows you no longer have."
|
|
312
|
+
),
|
|
313
|
+
)
|
|
314
|
+
)
|
|
315
|
+
elif texts and all(_ISO_8601.match(v) for v in texts):
|
|
316
|
+
notes.append(
|
|
317
|
+
Note(
|
|
318
|
+
severity="review",
|
|
319
|
+
entity=entity,
|
|
320
|
+
field=name,
|
|
321
|
+
what=(
|
|
322
|
+
"every sampled value looks like an ISO-8601 timestamp, and this is still "
|
|
323
|
+
"read as a string. Declare timestamptz if that is what it is; inferring it "
|
|
324
|
+
"would be right until the first value that is not."
|
|
325
|
+
),
|
|
326
|
+
)
|
|
327
|
+
)
|
|
328
|
+
return "string", notes
|
|
329
|
+
|
|
330
|
+
mapped = _FROM_TYPE.get(kind)
|
|
331
|
+
if mapped is None:
|
|
332
|
+
return "", [
|
|
333
|
+
Note(
|
|
334
|
+
severity="blocking",
|
|
335
|
+
entity=entity,
|
|
336
|
+
field=name,
|
|
337
|
+
what=(
|
|
338
|
+
f"{kind.__name__} has no neutral type. Convert it in your application, or "
|
|
339
|
+
f"declare the field yourself - a type nobody mapped cannot be stored in any "
|
|
340
|
+
f"engine, and a fallback to text would store it and lose it."
|
|
341
|
+
),
|
|
342
|
+
)
|
|
343
|
+
]
|
|
344
|
+
return mapped, notes
|
|
345
|
+
|
|
346
|
+
|
|
347
|
+
def infer_model(
|
|
348
|
+
rows: Iterable[Mapping[str, Any]], *, entity: str, sample_limit: int = 1000
|
|
349
|
+
) -> InferredModel:
|
|
350
|
+
"""Suggest a declaration for one entity from sample rows. Nothing leaves this process.
|
|
351
|
+
|
|
352
|
+
``sample_limit`` caps how many rows are read, because the point is a suggestion and reading a
|
|
353
|
+
hundred million rows to produce one is a cost with no return. The number actually used is
|
|
354
|
+
recorded in the result, since every claim in it is conditional on that number.
|
|
355
|
+
|
|
356
|
+
Raises on no rows at all rather than returning an empty declaration: an empty model builds
|
|
357
|
+
without error, places nothing, and the failure surfaces as a placement map with no groups.
|
|
358
|
+
"""
|
|
359
|
+
sampled: list[Mapping[str, Any]] = []
|
|
360
|
+
for row in rows:
|
|
361
|
+
if len(sampled) >= sample_limit:
|
|
362
|
+
break
|
|
363
|
+
if not isinstance(row, Mapping):
|
|
364
|
+
raise DeclarationError(
|
|
365
|
+
f"a sample row is {type(row).__name__}, not a mapping of column to value. "
|
|
366
|
+
f"Inference reads column names, so a positional row has nothing to name."
|
|
367
|
+
)
|
|
368
|
+
sampled.append(row)
|
|
369
|
+
|
|
370
|
+
if not sampled:
|
|
371
|
+
raise DeclarationError(
|
|
372
|
+
"no sample rows. An empty declaration builds without error, places nothing, and the "
|
|
373
|
+
"failure would surface later as a placement map with no groups."
|
|
374
|
+
)
|
|
375
|
+
|
|
376
|
+
columns: dict[str, list[Any]] = {}
|
|
377
|
+
for row in sampled:
|
|
378
|
+
for name, value in row.items():
|
|
379
|
+
columns.setdefault(str(name), []).append(value)
|
|
380
|
+
|
|
381
|
+
notes: list[Note] = []
|
|
382
|
+
fields: list[dict[str, str]] = []
|
|
383
|
+
for name in sorted(columns):
|
|
384
|
+
values = columns[name]
|
|
385
|
+
if len(values) < len(sampled):
|
|
386
|
+
notes.append(
|
|
387
|
+
Note(
|
|
388
|
+
severity="review",
|
|
389
|
+
entity=entity,
|
|
390
|
+
field=name,
|
|
391
|
+
what=(
|
|
392
|
+
f"present in {len(values)} of {len(sampled)} sampled rows. Every "
|
|
393
|
+
f"field in a declaration is required; if this one is genuinely "
|
|
394
|
+
f"optional, that is a modelling decision rather than a type."
|
|
395
|
+
),
|
|
396
|
+
)
|
|
397
|
+
)
|
|
398
|
+
inferred, field_notes = _neutral_type(entity, name, values)
|
|
399
|
+
notes.extend(field_notes)
|
|
400
|
+
if inferred:
|
|
401
|
+
fields.append({"name": name, "type": inferred})
|
|
402
|
+
|
|
403
|
+
candidates = tuple(
|
|
404
|
+
name
|
|
405
|
+
for name in sorted(columns)
|
|
406
|
+
if len(columns[name]) == len(sampled)
|
|
407
|
+
and None not in columns[name]
|
|
408
|
+
and len({_hashable(v) for v in columns[name]}) == len(sampled)
|
|
409
|
+
)
|
|
410
|
+
|
|
411
|
+
declaration: dict[str, Any] = {
|
|
412
|
+
"entities": [{"name": entity, "fields": fields}],
|
|
413
|
+
"relations": [],
|
|
414
|
+
"atomic": [],
|
|
415
|
+
}
|
|
416
|
+
|
|
417
|
+
return InferredModel(
|
|
418
|
+
declaration=declaration,
|
|
419
|
+
sample_size=len(sampled),
|
|
420
|
+
candidate_keys={entity: candidates},
|
|
421
|
+
notes=tuple(notes),
|
|
422
|
+
)
|
|
423
|
+
|
|
424
|
+
|
|
425
|
+
def _hashable(value: Any) -> Any:
|
|
426
|
+
"""Values as something a set can hold, so uniqueness can be counted at all."""
|
|
427
|
+
if isinstance(value, (list, dict, set)):
|
|
428
|
+
return repr(value)
|
|
429
|
+
return value
|
|
430
|
+
|
|
431
|
+
|
|
432
|
+
def infer_models(
|
|
433
|
+
named_rows: Mapping[str, Iterable[Mapping[str, Any]]], *, sample_limit: int = 1000
|
|
434
|
+
) -> InferredModel:
|
|
435
|
+
"""Several entities in one declaration, which is what a real application has.
|
|
436
|
+
|
|
437
|
+
Relations are **not** inferred. A column in one entity holding values that appear as another
|
|
438
|
+
entity's key looks exactly like a foreign key and looks exactly the same when it is a
|
|
439
|
+
coincidence of two independent identifier spaces - and a relation is what merges two entities
|
|
440
|
+
into one colocation group, so guessing one wrong changes which engine both of them live in.
|
|
441
|
+
"""
|
|
442
|
+
entities: list[dict[str, Any]] = []
|
|
443
|
+
candidates: dict[str, tuple[str, ...]] = {}
|
|
444
|
+
notes: list[Note] = []
|
|
445
|
+
sizes: list[int] = []
|
|
446
|
+
|
|
447
|
+
for entity in sorted(named_rows):
|
|
448
|
+
one = infer_model(named_rows[entity], entity=entity, sample_limit=sample_limit)
|
|
449
|
+
entities.extend(one.declaration["entities"])
|
|
450
|
+
candidates.update(one.candidate_keys)
|
|
451
|
+
notes.extend(one.notes)
|
|
452
|
+
sizes.append(one.sample_size)
|
|
453
|
+
|
|
454
|
+
return InferredModel(
|
|
455
|
+
declaration={"entities": entities, "relations": [], "atomic": []},
|
|
456
|
+
# The smallest sample, because every claim in the result is only as good as the weakest one
|
|
457
|
+
# behind it and a mean would hide an entity inferred from two rows.
|
|
458
|
+
sample_size=min(sizes),
|
|
459
|
+
candidate_keys=candidates,
|
|
460
|
+
notes=tuple(notes),
|
|
461
|
+
)
|
sde/internal.py
ADDED
|
@@ -0,0 +1,90 @@
|
|
|
1
|
+
"""The boundary between our bugs and the client's uptime.
|
|
2
|
+
|
|
3
|
+
This library runs inside somebody else's application. A defect in our telemetry, our logging or our
|
|
4
|
+
diagnostics must not take down their request, and the honest way to guarantee that is to route every
|
|
5
|
+
purely-internal side effect through one function that cannot raise.
|
|
6
|
+
|
|
7
|
+
The hard part is not the try/except. It is deciding what counts as internal, and getting that wrong
|
|
8
|
+
in either direction is bad:
|
|
9
|
+
|
|
10
|
+
- Swallow too little and a bug in a counter becomes a customer's outage.
|
|
11
|
+
- Swallow too much and a write that never happened is reported as success, which is the worst thing
|
|
12
|
+
this library could do.
|
|
13
|
+
|
|
14
|
+
So the rule is narrow and stated once here: **internal means it cannot change whether the client's
|
|
15
|
+
operation was performed correctly.** Emitting a log line is internal. Aggregating a telemetry window
|
|
16
|
+
is internal. Choosing which materialisation a read goes to is *not* - a wrong route returns wrong
|
|
17
|
+
data. Writing a row is not. Verifying a placement map's signature is not, because the map decides
|
|
18
|
+
where data is written.
|
|
19
|
+
|
|
20
|
+
When a guarded operation does fail, it is logged as ``sde.internal.error`` and the failure is
|
|
21
|
+
counted. A library that swallows silently is indistinguishable from one that works, so the counter
|
|
22
|
+
is the thing that makes this honest: it is readable, and a client who wants to alert on "the
|
|
23
|
+
vendor's library is quietly failing" has something to alert on.
|
|
24
|
+
"""
|
|
25
|
+
|
|
26
|
+
from __future__ import annotations
|
|
27
|
+
|
|
28
|
+
import threading
|
|
29
|
+
from collections.abc import Callable
|
|
30
|
+
from typing import TypeVar
|
|
31
|
+
|
|
32
|
+
__all__ = ["guard", "internal_failures", "reset_internal_failures"]
|
|
33
|
+
|
|
34
|
+
T = TypeVar("T")
|
|
35
|
+
|
|
36
|
+
_lock = threading.Lock()
|
|
37
|
+
_failures: dict[str, int] = {}
|
|
38
|
+
|
|
39
|
+
|
|
40
|
+
def internal_failures() -> dict[str, int]:
|
|
41
|
+
"""How many times each guarded operation has failed, by name.
|
|
42
|
+
|
|
43
|
+
Exposed rather than hidden. Swallowing a failure and leaving no trace of it would make this
|
|
44
|
+
library indistinguishable from one that works, and a client should be able to see that our code
|
|
45
|
+
is misbehaving inside their process even when we cannot.
|
|
46
|
+
"""
|
|
47
|
+
with _lock:
|
|
48
|
+
return dict(_failures)
|
|
49
|
+
|
|
50
|
+
|
|
51
|
+
def reset_internal_failures() -> None:
|
|
52
|
+
"""For tests."""
|
|
53
|
+
with _lock:
|
|
54
|
+
_failures.clear()
|
|
55
|
+
|
|
56
|
+
|
|
57
|
+
def guard(what: str, operation: Callable[[], T]) -> T | None:
|
|
58
|
+
"""Run a purely-internal operation. Never raises.
|
|
59
|
+
|
|
60
|
+
``what`` names the operation and becomes the key in :func:`internal_failures`, so it should be
|
|
61
|
+
stable across releases - a client may be alerting on it.
|
|
62
|
+
"""
|
|
63
|
+
try:
|
|
64
|
+
return operation()
|
|
65
|
+
except BaseException as exc:
|
|
66
|
+
# BaseException rather than Exception on purpose. A generator misbehaving, a recursion limit
|
|
67
|
+
# or a bad __del__ inside a telemetry aggregator would otherwise escape a narrower clause
|
|
68
|
+
# and reach the client's code, which is exactly what this exists to prevent.
|
|
69
|
+
# KeyboardInterrupt and SystemExit are re-raised below, because swallowing those would make
|
|
70
|
+
# an application impossible to stop.
|
|
71
|
+
if isinstance(exc, (KeyboardInterrupt, SystemExit)):
|
|
72
|
+
raise
|
|
73
|
+
with _lock:
|
|
74
|
+
_failures[what] = _failures.get(what, 0) + 1
|
|
75
|
+
_report(what, exc)
|
|
76
|
+
return None
|
|
77
|
+
|
|
78
|
+
|
|
79
|
+
def _report(what: str, exc: BaseException) -> None:
|
|
80
|
+
"""Log the failure, and do not let logging failures escape either.
|
|
81
|
+
|
|
82
|
+
Nested guarding sounds paranoid until you consider what a broken logging handler in a client's
|
|
83
|
+
application does to a library that logs from inside an exception handler.
|
|
84
|
+
"""
|
|
85
|
+
try:
|
|
86
|
+
from .logging import log
|
|
87
|
+
|
|
88
|
+
log("sde.internal.error", operation=what, error=type(exc).__name__)
|
|
89
|
+
except BaseException:
|
|
90
|
+
pass
|