smart-data-engine-sdk 0.1.0.dev0__py3-none-any.whl
This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
- sde/__init__.py +226 -0
- sde/canonical.py +141 -0
- sde/capabilities.py +62 -0
- sde/engines/__init__.py +0 -0
- sde/engines/clickhouse.py +689 -0
- sde/engines/orderbook.py +454 -0
- sde/engines/postgres.py +672 -0
- sde/entity.py +170 -0
- sde/errors.py +88 -0
- sde/explain.py +300 -0
- sde/groups.py +97 -0
- sde/hashing.py +242 -0
- sde/infer.py +461 -0
- sde/internal.py +90 -0
- sde/layout.py +660 -0
- sde/logging.py +132 -0
- sde/migration.py +820 -0
- sde/model.py +482 -0
- sde/placement.py +818 -0
- sde/py.typed +0 -0
- sde/routing.py +85 -0
- sde/schema.py +370 -0
- sde/session.py +507 -0
- sde/shapes.py +153 -0
- sde/telemetry.py +736 -0
- sde/testing/__init__.py +14 -0
- sde/testing/loader.py +175 -0
- sde/testing/memory.py +318 -0
- sde/types.py +228 -0
- sde/watermark.py +222 -0
- smart_data_engine_sdk-0.1.0.dev0.dist-info/METADATA +152 -0
- smart_data_engine_sdk-0.1.0.dev0.dist-info/RECORD +35 -0
- smart_data_engine_sdk-0.1.0.dev0.dist-info/WHEEL +4 -0
- smart_data_engine_sdk-0.1.0.dev0.dist-info/licenses/LICENSE +201 -0
- smart_data_engine_sdk-0.1.0.dev0.dist-info/licenses/NOTICE +13 -0
sde/hashing.py
ADDED
|
@@ -0,0 +1,242 @@
|
|
|
1
|
+
"""Hiding identifier names from us, and from the model, without hiding them from the planner's job.
|
|
2
|
+
|
|
3
|
+
A client in a regulated environment may not want us to see that they have a table called
|
|
4
|
+
``patient_diagnosis``. Requirement 11.3 lets them hash entity, field and relation names with a salt
|
|
5
|
+
that never leaves their infrastructure, and requirement 11.4 says the consequence must not be a
|
|
6
|
+
worse placement: no scoring feature may read the *meaning* of a name, so a hashed model has to be
|
|
7
|
+
placed identically to an unhashed one.
|
|
8
|
+
|
|
9
|
+
That second requirement is the interesting one. It is enforced by a test rather than by review, and
|
|
10
|
+
the test is worth more than a dozen unit tests of the planner, because it fails the moment anyone
|
|
11
|
+
writes ``if "log" in table_name`` anywhere in the decision path.
|
|
12
|
+
|
|
13
|
+
# Two consequences worth knowing before turning this on
|
|
14
|
+
|
|
15
|
+
**Hashing is a model change.** A group's name is its alphabetically first member, and hashing
|
|
16
|
+
changes which member that is; operation shape identifiers include the group and entity names, so
|
|
17
|
+
they change too. The canonical IR is different, so ``model_version`` is different, so the placement
|
|
18
|
+
map is different. Switching hashing on or off is therefore not a setting - it is a new model that
|
|
19
|
+
needs a new map, exactly as requirement 1.3 describes. Saying this plainly is cheaper than having a
|
|
20
|
+
client discover it when their existing map is refused.
|
|
21
|
+
|
|
22
|
+
**Your own tables get opaque names.** The physical layout comes from the map, which is keyed by
|
|
23
|
+
hashed names, so a table in the client's own database ends up called ``e_9c1f2a7b3d40``. For a
|
|
24
|
+
regulated deployment that is the point. For most deployments it is a real cost, and it should be
|
|
25
|
+
weighed rather than accepted by default - which is why this is off unless asked for.
|
|
26
|
+
|
|
27
|
+
# What is not hidden
|
|
28
|
+
|
|
29
|
+
Types, cardinalities, latencies, call counts, the shape of the graph. All of that is what the
|
|
30
|
+
planner actually reasons from, and none of it is a name. What the model loses is semantic signal:
|
|
31
|
+
``orders`` tells a language model something that ``e_9c1f2a7b3d40`` does not, so proposals get
|
|
32
|
+
measurably worse (requirement 18.10). The deterministic path is unaffected, which is the whole point
|
|
33
|
+
of it not reading names.
|
|
34
|
+
"""
|
|
35
|
+
|
|
36
|
+
from __future__ import annotations
|
|
37
|
+
|
|
38
|
+
import hashlib
|
|
39
|
+
import hmac
|
|
40
|
+
import os
|
|
41
|
+
import secrets
|
|
42
|
+
import unicodedata
|
|
43
|
+
from collections.abc import Mapping
|
|
44
|
+
from dataclasses import dataclass
|
|
45
|
+
from pathlib import Path
|
|
46
|
+
|
|
47
|
+
from .errors import DeclarationError
|
|
48
|
+
from .model import EntitySpec, FieldSpec, LogicalModel, RelationSpec, assemble
|
|
49
|
+
|
|
50
|
+
__all__ = ["NameMap", "hash_identifiers", "load_or_create_salt"]
|
|
51
|
+
|
|
52
|
+
# Twelve hex characters is 48 bits. Collisions matter here - two entities hashing to one name would
|
|
53
|
+
# merge them in the IR - so this is checked rather than assumed: hash_identifiers refuses on
|
|
54
|
+
# collision instead of producing a model with one entity where there were two.
|
|
55
|
+
DIGEST_CHARS = 12
|
|
56
|
+
|
|
57
|
+
_ENTITY_PREFIX = "e_"
|
|
58
|
+
_FIELD_PREFIX = "f_"
|
|
59
|
+
_RELATION_PREFIX = "r_"
|
|
60
|
+
|
|
61
|
+
|
|
62
|
+
def load_or_create_salt(path: Path | None = None) -> bytes:
|
|
63
|
+
"""Read the client's salt, generating it on first use.
|
|
64
|
+
|
|
65
|
+
The salt never appears in a model, a telemetry window or anything sent to us: it is the whole
|
|
66
|
+
mechanism, and a salt we hold is a mechanism we could reverse. Stored with owner-only
|
|
67
|
+
permissions, because a readable salt in a shared container image is the same as no salt.
|
|
68
|
+
|
|
69
|
+
The file is read **verbatim**. An earlier version called ``.strip()`` on it, to be forgiving
|
|
70
|
+
about a trailing newline in a hand-made file, and that was a serious bug rather than a kindness.
|
|
71
|
+
``bytes.strip()`` removes six byte *values* - space, tab, newline, carriage return, ``\x0b`` and
|
|
72
|
+
``\x0c`` - and a random 32-byte salt begins or ends with one of them about 5% of the time.
|
|
73
|
+
|
|
74
|
+
When it did, the process that generated the salt used all 32 bytes and every process afterwards
|
|
75
|
+
used the stripped remainder. So one client computed **two different model versions** for one
|
|
76
|
+
declared model: the map issued for one is refused by the other, and a map that was accepted
|
|
77
|
+
names different tables. The TypeScript library takes the salt as bytes from its caller and
|
|
78
|
+
strips nothing, so a Python service and a Node service sharing one salt file disagreed too - the
|
|
79
|
+
exact failure the byte contract exists to prevent, arriving through the file rather than the
|
|
80
|
+
encoder.
|
|
81
|
+
|
|
82
|
+
A file created with ``echo`` therefore includes its newline, and the salt is those bytes
|
|
83
|
+
including it. That is consistent and checkable. Being forgiving would mean choosing an encoding,
|
|
84
|
+
hex or base64, and that changes what is on disk - which changes every name derived from it.
|
|
85
|
+
"""
|
|
86
|
+
location = path or Path.home() / ".sde" / "salt"
|
|
87
|
+
if location.exists():
|
|
88
|
+
salt = location.read_bytes()
|
|
89
|
+
if len(salt) < 16:
|
|
90
|
+
raise DeclarationError(
|
|
91
|
+
f"the salt at {location} is shorter than 16 bytes. A short salt is "
|
|
92
|
+
"guessable, and a guessable salt means the names are not hidden. Delete the file "
|
|
93
|
+
"to have a new one generated - but note that a new salt is a new model version "
|
|
94
|
+
"and needs a new map."
|
|
95
|
+
)
|
|
96
|
+
return salt
|
|
97
|
+
|
|
98
|
+
location.parent.mkdir(parents=True, exist_ok=True)
|
|
99
|
+
salt = secrets.token_bytes(32)
|
|
100
|
+
# Written before the mode is set, then narrowed. os.open with 0o600 would be tighter, and is
|
|
101
|
+
# what this should become if the file ever holds more than a salt.
|
|
102
|
+
location.write_bytes(salt)
|
|
103
|
+
os.chmod(location, 0o600)
|
|
104
|
+
return salt
|
|
105
|
+
|
|
106
|
+
|
|
107
|
+
@dataclass(frozen=True)
|
|
108
|
+
class NameMap:
|
|
109
|
+
"""The translation between what the client wrote and what we see.
|
|
110
|
+
|
|
111
|
+
Held only in the client's process. The library needs it because the placement map is keyed by
|
|
112
|
+
hashed names while the application still says ``session.save("User", ...)`` - so every lookup
|
|
113
|
+
crosses this boundary, and it is the only place that does.
|
|
114
|
+
"""
|
|
115
|
+
|
|
116
|
+
entities: Mapping[str, str]
|
|
117
|
+
fields: Mapping[str, Mapping[str, str]]
|
|
118
|
+
relations: Mapping[str, Mapping[str, str]]
|
|
119
|
+
|
|
120
|
+
def entity(self, name: str) -> str:
|
|
121
|
+
try:
|
|
122
|
+
return self.entities[name]
|
|
123
|
+
except KeyError:
|
|
124
|
+
raise DeclarationError(
|
|
125
|
+
f"{name!r} is not in this model, so it has no hashed name. Either it was never "
|
|
126
|
+
"declared, or the model was rebuilt without it."
|
|
127
|
+
) from None
|
|
128
|
+
|
|
129
|
+
def field(self, entity: str, name: str) -> str:
|
|
130
|
+
return self.fields[entity][name]
|
|
131
|
+
|
|
132
|
+
def relation(self, entity: str, name: str) -> str:
|
|
133
|
+
return self.relations[entity][name]
|
|
134
|
+
|
|
135
|
+
|
|
136
|
+
def _digest(salt: bytes, prefix: str, *parts: str) -> str:
|
|
137
|
+
"""HMAC over the parts, NFC-normalised and joined by a separator no identifier can contain.
|
|
138
|
+
|
|
139
|
+
Two details, both of which a port can get wrong silently, so both are pinned in the format
|
|
140
|
+
contract.
|
|
141
|
+
|
|
142
|
+
**NFC first.** The canonical encoder normalises before it emits bytes, which is why two
|
|
143
|
+
libraries that declare ``Zamówienie`` in different normal forms compute the *same* model
|
|
144
|
+
version. Hashing a name before normalising it would throw that away: the same identifier written
|
|
145
|
+
two ways would give two digests, two model versions, and a placement map issued for one service
|
|
146
|
+
that the other refuses. Nothing about that is visible in ASCII, so it would have shipped and
|
|
147
|
+
then failed for a client whose entity names are not English.
|
|
148
|
+
|
|
149
|
+
**The separator, not concatenation.** Hashing ``("User", "id")`` as ``"Userid"`` would collide
|
|
150
|
+
with ``("Use", "rid")``. Unlikely is not a guarantee when the consequence is two fields becoming
|
|
151
|
+
one column. U+0000 cannot occur in an identifier, so the join is unambiguous.
|
|
152
|
+
|
|
153
|
+
The prefix is deliberately outside the HMAC. It labels the digest for a human reading a table
|
|
154
|
+
name; it carries no secret and adding it to the message would only make the derivation harder to
|
|
155
|
+
reproduce.
|
|
156
|
+
"""
|
|
157
|
+
message = "\x00".join(unicodedata.normalize("NFC", part) for part in parts)
|
|
158
|
+
digest = hmac.new(salt, message.encode("utf-8"), hashlib.sha256).hexdigest()
|
|
159
|
+
return prefix + digest[:DIGEST_CHARS]
|
|
160
|
+
|
|
161
|
+
|
|
162
|
+
def hash_identifiers(model: LogicalModel, salt: bytes) -> tuple[LogicalModel, NameMap]:
|
|
163
|
+
"""Return an equivalent model with every identifier replaced by a keyed digest.
|
|
164
|
+
|
|
165
|
+
Field and relation names are hashed *with their entity* in the input, so the same field name on
|
|
166
|
+
two entities produces two different digests. That is not paranoia: leaving them independent
|
|
167
|
+
would let us learn that two entities share a field called ``email`` even though we cannot read
|
|
168
|
+
the name, which is exactly the kind of structural leak hashing is meant to close.
|
|
169
|
+
"""
|
|
170
|
+
if len(salt) < 16:
|
|
171
|
+
raise DeclarationError("the salt must be at least 16 bytes")
|
|
172
|
+
|
|
173
|
+
entity_names: dict[str, str] = {}
|
|
174
|
+
for spec in model.entities:
|
|
175
|
+
hashed = _digest(salt, _ENTITY_PREFIX, spec.name)
|
|
176
|
+
if hashed in entity_names.values():
|
|
177
|
+
clash = next(k for k, v in entity_names.items() if v == hashed)
|
|
178
|
+
raise DeclarationError(
|
|
179
|
+
f"{spec.name!r} and {clash!r} hash to the same name. Refused rather than merged: a "
|
|
180
|
+
"model with one entity where there were two would place both in one engine and "
|
|
181
|
+
"write both into one table. Change the salt."
|
|
182
|
+
)
|
|
183
|
+
entity_names[spec.name] = hashed
|
|
184
|
+
|
|
185
|
+
field_names: dict[str, dict[str, str]] = {}
|
|
186
|
+
relation_names: dict[str, dict[str, str]] = {}
|
|
187
|
+
|
|
188
|
+
entities: list[EntitySpec] = []
|
|
189
|
+
for spec in model.entities:
|
|
190
|
+
mapping: dict[str, str] = {}
|
|
191
|
+
for spec_field in spec.fields:
|
|
192
|
+
mapping[spec_field.name] = _digest(
|
|
193
|
+
salt, _FIELD_PREFIX, spec.name, spec_field.name
|
|
194
|
+
)
|
|
195
|
+
if len(set(mapping.values())) != len(mapping):
|
|
196
|
+
raise DeclarationError(
|
|
197
|
+
f"two fields of {spec.name!r} hash to the same name. Refused rather than merged. "
|
|
198
|
+
"Change the salt."
|
|
199
|
+
)
|
|
200
|
+
field_names[spec.name] = mapping
|
|
201
|
+
|
|
202
|
+
entities.append(
|
|
203
|
+
EntitySpec(
|
|
204
|
+
name=entity_names[spec.name],
|
|
205
|
+
fields=tuple(
|
|
206
|
+
FieldSpec(name=mapping[f.name], type=f.type, nullable=f.nullable)
|
|
207
|
+
for f in spec.fields
|
|
208
|
+
),
|
|
209
|
+
key=tuple(mapping[k] for k in spec.key),
|
|
210
|
+
pii=tuple(mapping[p] for p in spec.pii),
|
|
211
|
+
# Residency is not an identifier. It is a jurisdiction, it is a hard constraint on
|
|
212
|
+
# placement, and hashing it would make the constraint unenforceable.
|
|
213
|
+
residency=spec.residency,
|
|
214
|
+
)
|
|
215
|
+
)
|
|
216
|
+
|
|
217
|
+
relations: list[RelationSpec] = []
|
|
218
|
+
for relation in model.relations:
|
|
219
|
+
hashed = _digest(salt, _RELATION_PREFIX, relation.source, relation.name)
|
|
220
|
+
relation_names.setdefault(relation.source, {})[relation.name] = hashed
|
|
221
|
+
relations.append(
|
|
222
|
+
RelationSpec(
|
|
223
|
+
name=hashed,
|
|
224
|
+
source=entity_names[relation.source],
|
|
225
|
+
target=entity_names[relation.target],
|
|
226
|
+
)
|
|
227
|
+
)
|
|
228
|
+
|
|
229
|
+
atomic = tuple(
|
|
230
|
+
tuple(sorted(entity_names[member] for member in group)) for group in model.atomic
|
|
231
|
+
)
|
|
232
|
+
|
|
233
|
+
hashed_model = assemble(
|
|
234
|
+
entities=tuple(entities),
|
|
235
|
+
relations=tuple(relations),
|
|
236
|
+
atomic=tuple(sorted(atomic)),
|
|
237
|
+
# The cost ceiling is a number and a currency, not an identifier.
|
|
238
|
+
cost_ceiling=model.cost_ceiling,
|
|
239
|
+
)
|
|
240
|
+
return hashed_model, NameMap(
|
|
241
|
+
entities=entity_names, fields=field_names, relations=relation_names
|
|
242
|
+
)
|