smart-data-engine-sdk 0.1.0.dev0__py3-none-any.whl

This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
sde/hashing.py ADDED
@@ -0,0 +1,242 @@
1
+ """Hiding identifier names from us, and from the model, without hiding them from the planner's job.
2
+
3
+ A client in a regulated environment may not want us to see that they have a table called
4
+ ``patient_diagnosis``. Requirement 11.3 lets them hash entity, field and relation names with a salt
5
+ that never leaves their infrastructure, and requirement 11.4 says the consequence must not be a
6
+ worse placement: no scoring feature may read the *meaning* of a name, so a hashed model has to be
7
+ placed identically to an unhashed one.
8
+
9
+ That second requirement is the interesting one. It is enforced by a test rather than by review, and
10
+ the test is worth more than a dozen unit tests of the planner, because it fails the moment anyone
11
+ writes ``if "log" in table_name`` anywhere in the decision path.
12
+
13
+ # Two consequences worth knowing before turning this on
14
+
15
+ **Hashing is a model change.** A group's name is its alphabetically first member, and hashing
16
+ changes which member that is; operation shape identifiers include the group and entity names, so
17
+ they change too. The canonical IR is different, so ``model_version`` is different, so the placement
18
+ map is different. Switching hashing on or off is therefore not a setting - it is a new model that
19
+ needs a new map, exactly as requirement 1.3 describes. Saying this plainly is cheaper than having a
20
+ client discover it when their existing map is refused.
21
+
22
+ **Your own tables get opaque names.** The physical layout comes from the map, which is keyed by
23
+ hashed names, so a table in the client's own database ends up called ``e_9c1f2a7b3d40``. For a
24
+ regulated deployment that is the point. For most deployments it is a real cost, and it should be
25
+ weighed rather than accepted by default - which is why this is off unless asked for.
26
+
27
+ # What is not hidden
28
+
29
+ Types, cardinalities, latencies, call counts, the shape of the graph. All of that is what the
30
+ planner actually reasons from, and none of it is a name. What the model loses is semantic signal:
31
+ ``orders`` tells a language model something that ``e_9c1f2a7b3d40`` does not, so proposals get
32
+ measurably worse (requirement 18.10). The deterministic path is unaffected, which is the whole point
33
+ of it not reading names.
34
+ """
35
+
36
+ from __future__ import annotations
37
+
38
+ import hashlib
39
+ import hmac
40
+ import os
41
+ import secrets
42
+ import unicodedata
43
+ from collections.abc import Mapping
44
+ from dataclasses import dataclass
45
+ from pathlib import Path
46
+
47
+ from .errors import DeclarationError
48
+ from .model import EntitySpec, FieldSpec, LogicalModel, RelationSpec, assemble
49
+
50
+ __all__ = ["NameMap", "hash_identifiers", "load_or_create_salt"]
51
+
52
+ # Twelve hex characters is 48 bits. Collisions matter here - two entities hashing to one name would
53
+ # merge them in the IR - so this is checked rather than assumed: hash_identifiers refuses on
54
+ # collision instead of producing a model with one entity where there were two.
55
+ DIGEST_CHARS = 12
56
+
57
+ _ENTITY_PREFIX = "e_"
58
+ _FIELD_PREFIX = "f_"
59
+ _RELATION_PREFIX = "r_"
60
+
61
+
62
+ def load_or_create_salt(path: Path | None = None) -> bytes:
63
+ """Read the client's salt, generating it on first use.
64
+
65
+ The salt never appears in a model, a telemetry window or anything sent to us: it is the whole
66
+ mechanism, and a salt we hold is a mechanism we could reverse. Stored with owner-only
67
+ permissions, because a readable salt in a shared container image is the same as no salt.
68
+
69
+ The file is read **verbatim**. An earlier version called ``.strip()`` on it, to be forgiving
70
+ about a trailing newline in a hand-made file, and that was a serious bug rather than a kindness.
71
+ ``bytes.strip()`` removes six byte *values* - space, tab, newline, carriage return, ``\x0b`` and
72
+ ``\x0c`` - and a random 32-byte salt begins or ends with one of them about 5% of the time.
73
+
74
+ When it did, the process that generated the salt used all 32 bytes and every process afterwards
75
+ used the stripped remainder. So one client computed **two different model versions** for one
76
+ declared model: the map issued for one is refused by the other, and a map that was accepted
77
+ names different tables. The TypeScript library takes the salt as bytes from its caller and
78
+ strips nothing, so a Python service and a Node service sharing one salt file disagreed too - the
79
+ exact failure the byte contract exists to prevent, arriving through the file rather than the
80
+ encoder.
81
+
82
+ A file created with ``echo`` therefore includes its newline, and the salt is those bytes
83
+ including it. That is consistent and checkable. Being forgiving would mean choosing an encoding,
84
+ hex or base64, and that changes what is on disk - which changes every name derived from it.
85
+ """
86
+ location = path or Path.home() / ".sde" / "salt"
87
+ if location.exists():
88
+ salt = location.read_bytes()
89
+ if len(salt) < 16:
90
+ raise DeclarationError(
91
+ f"the salt at {location} is shorter than 16 bytes. A short salt is "
92
+ "guessable, and a guessable salt means the names are not hidden. Delete the file "
93
+ "to have a new one generated - but note that a new salt is a new model version "
94
+ "and needs a new map."
95
+ )
96
+ return salt
97
+
98
+ location.parent.mkdir(parents=True, exist_ok=True)
99
+ salt = secrets.token_bytes(32)
100
+ # Written before the mode is set, then narrowed. os.open with 0o600 would be tighter, and is
101
+ # what this should become if the file ever holds more than a salt.
102
+ location.write_bytes(salt)
103
+ os.chmod(location, 0o600)
104
+ return salt
105
+
106
+
107
+ @dataclass(frozen=True)
108
+ class NameMap:
109
+ """The translation between what the client wrote and what we see.
110
+
111
+ Held only in the client's process. The library needs it because the placement map is keyed by
112
+ hashed names while the application still says ``session.save("User", ...)`` - so every lookup
113
+ crosses this boundary, and it is the only place that does.
114
+ """
115
+
116
+ entities: Mapping[str, str]
117
+ fields: Mapping[str, Mapping[str, str]]
118
+ relations: Mapping[str, Mapping[str, str]]
119
+
120
+ def entity(self, name: str) -> str:
121
+ try:
122
+ return self.entities[name]
123
+ except KeyError:
124
+ raise DeclarationError(
125
+ f"{name!r} is not in this model, so it has no hashed name. Either it was never "
126
+ "declared, or the model was rebuilt without it."
127
+ ) from None
128
+
129
+ def field(self, entity: str, name: str) -> str:
130
+ return self.fields[entity][name]
131
+
132
+ def relation(self, entity: str, name: str) -> str:
133
+ return self.relations[entity][name]
134
+
135
+
136
+ def _digest(salt: bytes, prefix: str, *parts: str) -> str:
137
+ """HMAC over the parts, NFC-normalised and joined by a separator no identifier can contain.
138
+
139
+ Two details, both of which a port can get wrong silently, so both are pinned in the format
140
+ contract.
141
+
142
+ **NFC first.** The canonical encoder normalises before it emits bytes, which is why two
143
+ libraries that declare ``Zamówienie`` in different normal forms compute the *same* model
144
+ version. Hashing a name before normalising it would throw that away: the same identifier written
145
+ two ways would give two digests, two model versions, and a placement map issued for one service
146
+ that the other refuses. Nothing about that is visible in ASCII, so it would have shipped and
147
+ then failed for a client whose entity names are not English.
148
+
149
+ **The separator, not concatenation.** Hashing ``("User", "id")`` as ``"Userid"`` would collide
150
+ with ``("Use", "rid")``. Unlikely is not a guarantee when the consequence is two fields becoming
151
+ one column. U+0000 cannot occur in an identifier, so the join is unambiguous.
152
+
153
+ The prefix is deliberately outside the HMAC. It labels the digest for a human reading a table
154
+ name; it carries no secret and adding it to the message would only make the derivation harder to
155
+ reproduce.
156
+ """
157
+ message = "\x00".join(unicodedata.normalize("NFC", part) for part in parts)
158
+ digest = hmac.new(salt, message.encode("utf-8"), hashlib.sha256).hexdigest()
159
+ return prefix + digest[:DIGEST_CHARS]
160
+
161
+
162
+ def hash_identifiers(model: LogicalModel, salt: bytes) -> tuple[LogicalModel, NameMap]:
163
+ """Return an equivalent model with every identifier replaced by a keyed digest.
164
+
165
+ Field and relation names are hashed *with their entity* in the input, so the same field name on
166
+ two entities produces two different digests. That is not paranoia: leaving them independent
167
+ would let us learn that two entities share a field called ``email`` even though we cannot read
168
+ the name, which is exactly the kind of structural leak hashing is meant to close.
169
+ """
170
+ if len(salt) < 16:
171
+ raise DeclarationError("the salt must be at least 16 bytes")
172
+
173
+ entity_names: dict[str, str] = {}
174
+ for spec in model.entities:
175
+ hashed = _digest(salt, _ENTITY_PREFIX, spec.name)
176
+ if hashed in entity_names.values():
177
+ clash = next(k for k, v in entity_names.items() if v == hashed)
178
+ raise DeclarationError(
179
+ f"{spec.name!r} and {clash!r} hash to the same name. Refused rather than merged: a "
180
+ "model with one entity where there were two would place both in one engine and "
181
+ "write both into one table. Change the salt."
182
+ )
183
+ entity_names[spec.name] = hashed
184
+
185
+ field_names: dict[str, dict[str, str]] = {}
186
+ relation_names: dict[str, dict[str, str]] = {}
187
+
188
+ entities: list[EntitySpec] = []
189
+ for spec in model.entities:
190
+ mapping: dict[str, str] = {}
191
+ for spec_field in spec.fields:
192
+ mapping[spec_field.name] = _digest(
193
+ salt, _FIELD_PREFIX, spec.name, spec_field.name
194
+ )
195
+ if len(set(mapping.values())) != len(mapping):
196
+ raise DeclarationError(
197
+ f"two fields of {spec.name!r} hash to the same name. Refused rather than merged. "
198
+ "Change the salt."
199
+ )
200
+ field_names[spec.name] = mapping
201
+
202
+ entities.append(
203
+ EntitySpec(
204
+ name=entity_names[spec.name],
205
+ fields=tuple(
206
+ FieldSpec(name=mapping[f.name], type=f.type, nullable=f.nullable)
207
+ for f in spec.fields
208
+ ),
209
+ key=tuple(mapping[k] for k in spec.key),
210
+ pii=tuple(mapping[p] for p in spec.pii),
211
+ # Residency is not an identifier. It is a jurisdiction, it is a hard constraint on
212
+ # placement, and hashing it would make the constraint unenforceable.
213
+ residency=spec.residency,
214
+ )
215
+ )
216
+
217
+ relations: list[RelationSpec] = []
218
+ for relation in model.relations:
219
+ hashed = _digest(salt, _RELATION_PREFIX, relation.source, relation.name)
220
+ relation_names.setdefault(relation.source, {})[relation.name] = hashed
221
+ relations.append(
222
+ RelationSpec(
223
+ name=hashed,
224
+ source=entity_names[relation.source],
225
+ target=entity_names[relation.target],
226
+ )
227
+ )
228
+
229
+ atomic = tuple(
230
+ tuple(sorted(entity_names[member] for member in group)) for group in model.atomic
231
+ )
232
+
233
+ hashed_model = assemble(
234
+ entities=tuple(entities),
235
+ relations=tuple(relations),
236
+ atomic=tuple(sorted(atomic)),
237
+ # The cost ceiling is a number and a currency, not an identifier.
238
+ cost_ceiling=model.cost_ceiling,
239
+ )
240
+ return hashed_model, NameMap(
241
+ entities=entity_names, fields=field_names, relations=relation_names
242
+ )