smart-data-engine-sdk 0.1.0__py3-none-any.whl

This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
Files changed (73) hide show
  1. sde/__init__.py +318 -0
  2. sde/_cutover_project.py +179 -0
  3. sde/_local_state.py +188 -0
  4. sde/_operator_deadline.py +50 -0
  5. sde/_usage.py +314 -0
  6. sde/bulk.py +79 -0
  7. sde/canonical.py +141 -0
  8. sde/capabilities.py +62 -0
  9. sde/cutover.py +286 -0
  10. sde/engines/__init__.py +0 -0
  11. sde/engines/_clickhouse_connection.py +224 -0
  12. sde/engines/_index_build.py +294 -0
  13. sde/engines/_operator.py +394 -0
  14. sde/engines/_staging.py +222 -0
  15. sde/engines/_storage.py +22 -0
  16. sde/engines/_write_fences.py +271 -0
  17. sde/engines/clickhouse.py +1115 -0
  18. sde/engines/orderbook.py +457 -0
  19. sde/engines/postgres.py +967 -0
  20. sde/entity.py +170 -0
  21. sde/errors.py +103 -0
  22. sde/explain.py +300 -0
  23. sde/frozen_verification.py +152 -0
  24. sde/generation.py +131 -0
  25. sde/groups.py +97 -0
  26. sde/hashing.py +242 -0
  27. sde/index_build.py +313 -0
  28. sde/index_operator.py +347 -0
  29. sde/infer.py +461 -0
  30. sde/inspection.py +62 -0
  31. sde/internal.py +90 -0
  32. sde/layout.py +669 -0
  33. sde/local_cutover.py +801 -0
  34. sde/logging.py +143 -0
  35. sde/migration.py +856 -0
  36. sde/model.py +482 -0
  37. sde/physical.py +531 -0
  38. sde/placement.py +1010 -0
  39. sde/provisioning.py +63 -0
  40. sde/py.typed +0 -0
  41. sde/query.py +521 -0
  42. sde/routing.py +85 -0
  43. sde/schema.py +466 -0
  44. sde/session.py +993 -0
  45. sde/shapes.py +153 -0
  46. sde/staging.py +264 -0
  47. sde/staging_operator.py +393 -0
  48. sde/telemetry.py +1087 -0
  49. sde/testing/__init__.py +14 -0
  50. sde/testing/loader.py +175 -0
  51. sde/testing/memory.py +331 -0
  52. sde/types.py +228 -0
  53. sde/verification.py +220 -0
  54. sde/watermark.py +222 -0
  55. sde/write_fence.py +283 -0
  56. sde_demo/__init__.py +1 -0
  57. sde_demo/__main__.py +183 -0
  58. sde_demo/diagnostics.py +92 -0
  59. sde_demo/model.py +75 -0
  60. sde_demo/project.py +312 -0
  61. sde_demo/py.typed +0 -0
  62. sde_demo/query_count.py +301 -0
  63. sde_demo/resources.py +969 -0
  64. sde_demo/runtime.py +419 -0
  65. sde_demo/verification.py +242 -0
  66. sde_operator/__init__.py +1 -0
  67. sde_operator/__main__.py +210 -0
  68. smart_data_engine_sdk-0.1.0.dist-info/METADATA +174 -0
  69. smart_data_engine_sdk-0.1.0.dist-info/RECORD +73 -0
  70. smart_data_engine_sdk-0.1.0.dist-info/WHEEL +4 -0
  71. smart_data_engine_sdk-0.1.0.dist-info/entry_points.txt +3 -0
  72. smart_data_engine_sdk-0.1.0.dist-info/licenses/LICENSE +201 -0
  73. smart_data_engine_sdk-0.1.0.dist-info/licenses/NOTICE +13 -0
@@ -0,0 +1,152 @@
1
+ """Drain named native barriers and compare stable copies, including target-only rows.
2
+
3
+ This is a client-side data operation. It neither chooses a placement nor activates a map. It
4
+ leaves its barriers installed on success and failure; the executor owns the durable decision
5
+ which permits release, rollback or activation.
6
+ """
7
+
8
+ from __future__ import annotations
9
+
10
+ import re
11
+ from collections.abc import Mapping
12
+ from dataclasses import dataclass
13
+ from time import perf_counter_ns
14
+ from typing import Any, cast
15
+
16
+ from .errors import MigrationRefused
17
+ from .generation import Fencable, check_epoch
18
+ from .inspection import InspectionContext
19
+ from .migration import CHUNK_ROWS, VerifyReport, verify
20
+ from .verification import VerificationRequest
21
+ from .write_fence import FenceState, WriteFence
22
+
23
+
24
+ @dataclass(frozen=True)
25
+ class FrozenTable:
26
+ engine: str
27
+ materialization: str
28
+ table: str
29
+ identity: str
30
+ project_id: str
31
+ epoch: int
32
+ hold_id: str
33
+
34
+ def as_record(self) -> dict[str, Any]:
35
+ return {
36
+ "engine": self.engine,
37
+ "materialization": self.materialization,
38
+ "table": self.table,
39
+ "identity": self.identity,
40
+ "project_id": self.project_id,
41
+ "epoch": self.epoch,
42
+ "hold_id": self.hold_id,
43
+ }
44
+
45
+
46
+ @dataclass(frozen=True)
47
+ class FrozenVerifyReport:
48
+ comparison: VerifyReport
49
+ barriers: tuple[FrozenTable, ...]
50
+ elapsed_ms: int
51
+
52
+ @property
53
+ def matched(self) -> bool:
54
+ return (
55
+ self.comparison.matched and self.comparison.rows_source == self.comparison.rows_target
56
+ )
57
+
58
+ def as_record(self) -> dict[str, Any]:
59
+ return {
60
+ "protocol": 1,
61
+ "comparison": self.comparison.as_record(),
62
+ "barriers": [barrier.as_record() for barrier in self.barriers],
63
+ "elapsed_ms": self.elapsed_ms,
64
+ "matched": self.matched,
65
+ }
66
+
67
+
68
+ def _matches(state: FenceState, wanted: FrozenTable) -> None:
69
+ if state.identity != wanted.identity or state.project_id != wanted.project_id:
70
+ raise MigrationRefused("a frozen comparison table changed identity or project")
71
+ if not state.complete or state.epoch != wanted.epoch or wanted.hold_id not in state.holds:
72
+ raise MigrationRefused("a frozen comparison lost its named barrier or write generation")
73
+
74
+
75
+ def verify_frozen(
76
+ context: InspectionContext,
77
+ group: str,
78
+ *,
79
+ request: VerificationRequest,
80
+ hold_id: str,
81
+ epochs: Mapping[str, int],
82
+ chunk_rows: int = CHUNK_ROWS,
83
+ at: str | None = None,
84
+ ) -> FrozenVerifyReport:
85
+ """Acquire/drain each barrier and compare exact logical content; do not release the holds.
86
+
87
+ `epochs` names materialization ids, including the source. They can differ during operator
88
+ maintenance and therefore cannot be inferred from a runtime session's single group epoch.
89
+ A caller must budget this work and handle uncertain engine I/O as recovery, not as success.
90
+ """
91
+ if not isinstance(hold_id, str) or re.fullmatch("[0-9a-f]{32}", hold_id) is None:
92
+ raise MigrationRefused(
93
+ "a frozen comparison hold id must be 32 lowercase hexadecimal digits"
94
+ )
95
+ if chunk_rows < 1:
96
+ raise MigrationRefused("a frozen comparison needs a positive chunk size")
97
+ request.check_session(context.placement, project_id=context.project_id, group=group)
98
+ if at is not None:
99
+ request.check_time(at)
100
+ spot = context.placement.placement_of(group)
101
+ materials = (spot.source, *spot.also_write)
102
+ if not spot.also_write or set(epochs) != {material.id for material in materials}:
103
+ raise MigrationRefused("frozen comparison epochs must name exactly the source and copy ids")
104
+ checked_epochs = {name: check_epoch(value) for name, value in epochs.items()}
105
+ planned: list[tuple[WriteFence, FrozenTable]] = []
106
+ for material in sorted(materials, key=lambda value: (value.engine, value.id)):
107
+ engine = context.engines[material.engine]
108
+ if not callable(getattr(engine, "write_fence", None)) or not callable(
109
+ getattr(engine, "validate_schema", None)
110
+ ):
111
+ raise MigrationRefused(
112
+ "frozen comparison requires native write fences and schema checks"
113
+ )
114
+ native = cast(Fencable, engine)
115
+ native.validate_schema(material.layout)
116
+ for table in sorted(material.layout.tables.values()):
117
+ fence = native.write_fence(table, project_id=context.project_id)
118
+ state = fence.state()
119
+ epoch = checked_epochs[material.id]
120
+ if hold_id in state.retired:
121
+ raise MigrationRefused("a frozen comparison cannot reuse a retired barrier id")
122
+ if not state.complete or state.epoch != epoch:
123
+ raise MigrationRefused(
124
+ "a frozen comparison table is not at the expected write generation"
125
+ )
126
+ planned.append(
127
+ (
128
+ fence,
129
+ FrozenTable(
130
+ material.engine,
131
+ material.id,
132
+ table,
133
+ state.identity,
134
+ context.project_id,
135
+ epoch,
136
+ hold_id,
137
+ ),
138
+ )
139
+ )
140
+ started = perf_counter_ns()
141
+ for fence, wanted in planned:
142
+ # Constraint presence alone does not drain an old ClickHouse INSERT. Always execute the
143
+ # native drain, including when resuming an already installed hold.
144
+ _matches(fence.freeze(hold_id), wanted)
145
+ comparison = verify(context, group, request=request, chunk_rows=chunk_rows, at=at)
146
+ for fence, wanted in planned:
147
+ _matches(fence.state(), wanted)
148
+ return FrozenVerifyReport(
149
+ comparison,
150
+ tuple(wanted for _, wanted in planned),
151
+ (perf_counter_ns() - started) // 1_000_000,
152
+ )
sde/generation.py ADDED
@@ -0,0 +1,131 @@
1
+ """Wire vocabulary for generation-bearing maps, independent of drivers and placement classes."""
2
+
3
+ from __future__ import annotations
4
+
5
+ from collections.abc import Mapping, Sequence
6
+ from typing import TYPE_CHECKING, Any, Protocol, cast
7
+
8
+ from .errors import MigrationRefused
9
+
10
+ if TYPE_CHECKING:
11
+ from .model import LogicalModel
12
+ from .physical import PhysicalFinding
13
+ from .placement import PhysicalLayout, PlacementMap
14
+ from .session import Engine
15
+ from .write_fence import WriteFence
16
+
17
+
18
+ class Fencable(Protocol):
19
+ def write_fence(self, table: str, *, project_id: str) -> WriteFence: ...
20
+ def validate_schema(
21
+ self, layout: PhysicalLayout, *, keys: Mapping[str, Sequence[str]] | None = None
22
+ ) -> tuple[PhysicalFinding, ...]: ...
23
+
24
+
25
+ DRAIN_TABLE = "__sde_fence_drains"
26
+ EPOCH_COLUMN = "__sde_write_epoch"
27
+ MAX_EPOCH = 9_007_199_254_740_991
28
+ GENERATIONS_SINCE = 4
29
+ """The placement map contract that introduced ``project_id`` and per-group ``write_epoch``."""
30
+
31
+
32
+ def check_epoch(epoch: int) -> int:
33
+ if type(epoch) not in (int, float) or not 1 <= epoch <= MAX_EPOCH or int(epoch) != epoch:
34
+ raise MigrationRefused("write epoch must be a positive safe integer")
35
+ return int(epoch)
36
+
37
+
38
+ def json_numbers(value: Any) -> Any:
39
+ """JSON has one number type. Match integral JS numbers before checking/signing a v4 map.
40
+
41
+ Canonical encoding itself remains strict. Nonintegral/nonfinite numbers survive this pass and
42
+ are refused by the map's canonical boundary. Earlier map contracts retain their old behavior.
43
+ """
44
+ if isinstance(value, float) and value.is_integer():
45
+ return int(value)
46
+ if isinstance(value, dict):
47
+ return {key: json_numbers(item) for key, item in value.items()}
48
+ if isinstance(value, list):
49
+ return [json_numbers(item) for item in value]
50
+ if isinstance(value, tuple):
51
+ return tuple(json_numbers(item) for item in value)
52
+ return value
53
+
54
+
55
+ def check_map_project(placement: PlacementMap, project_id: str | None) -> str | None:
56
+ if placement.contract < 4:
57
+ return None
58
+ if project_id is None or project_id != placement.project_id:
59
+ raise MigrationRefused(
60
+ "this generation-bearing map needs its locally configured project_id; "
61
+ "do not learn that identity from the supplied map"
62
+ )
63
+ if placement.fingerprint is None:
64
+ raise MigrationRefused("generation-bearing sessions need an immutable loaded placement map")
65
+ return project_id
66
+
67
+
68
+ def validate_generations(
69
+ model: LogicalModel,
70
+ placement: PlacementMap,
71
+ engines: Mapping[str, Engine],
72
+ project_id: str | None,
73
+ ) -> tuple[PhysicalFinding, ...]:
74
+ """Check generations and columns; return how tables differ from the declared physical design.
75
+
76
+ The physical design is reported rather than refused: a running application must not stop
77
+ because a table's sort key or index differs from the map (requirement 3.6). Columns, types
78
+ and generations still refuse.
79
+ """
80
+ if placement.contract < 4:
81
+ return ()
82
+ findings: list[PhysicalFinding] = []
83
+ local_project = check_map_project(placement, project_id)
84
+ assert local_project is not None
85
+ if model.version != placement.model_version:
86
+ raise MigrationRefused("the generation-bearing map names another session model")
87
+ for name in sorted(placement.groups):
88
+ spot = placement.groups[name]
89
+ for material in spot.all():
90
+ factory = getattr(engines[material.engine], "write_fence", None)
91
+ if not callable(factory) or not callable(
92
+ getattr(engines[material.engine], "validate_schema", None)
93
+ ):
94
+ raise MigrationRefused(
95
+ f"engine {material.engine} does not implement write generations"
96
+ )
97
+ keys = {entity: model.entity(entity).key for entity in material.layout.tables}
98
+ findings.extend(
99
+ cast(Fencable, engines[material.engine]).validate_schema(
100
+ material.layout, keys=keys
101
+ )
102
+ )
103
+ for table in sorted(material.layout.tables.values()):
104
+ state = (
105
+ cast(Fencable, engines[material.engine])
106
+ .write_fence(table, project_id=local_project)
107
+ .state()
108
+ )
109
+ if not state.complete or state.epoch != spot.write_epoch:
110
+ raise MigrationRefused(
111
+ f"the write generation for {name} is not active in {material.engine}; "
112
+ "provision the signed map or load the current map before opening a session"
113
+ )
114
+ return tuple(findings)
115
+
116
+
117
+ def stamp_values(
118
+ placement: PlacementMap, group: str, values: Mapping[str, Any]
119
+ ) -> Mapping[str, Any]:
120
+ epoch = placement.placement_of(group).write_epoch
121
+ if epoch is None:
122
+ return values
123
+ if EPOCH_COLUMN in values:
124
+ raise MigrationRefused("the write-epoch column is reserved for the SDK")
125
+ return {**values, EPOCH_COLUMN: epoch}
126
+
127
+
128
+ def logical_row(placement: PlacementMap, row: Any) -> Any:
129
+ if placement.contract >= 4 and isinstance(row, dict):
130
+ return {key: value for key, value in row.items() if key != EPOCH_COLUMN}
131
+ return row
sde/groups.py ADDED
@@ -0,0 +1,97 @@
1
+ """Colocation groups: the unit of placement.
2
+
3
+ Entities that are queried together, or that must change together, live in the same engine. The graph
4
+ has an edge for every relation and for every declared atomicity, and a group is a connected
5
+ component of it.
6
+
7
+ This looks like a limitation and is the opposite. A join across two engines means pulling both sides
8
+ over the network and joining in the client's process: slow, memory-hungry, and the single easiest
9
+ way for a young product to embarrass itself. Making colocation a constraint turns that problem into
10
+ something the planner simply respects, and the consistency contract falls straight out of it - one
11
+ group, one engine, that engine's transaction semantics, and no distributed transactions anywhere.
12
+
13
+ The obvious worry is that every relation being an edge collapses a normalised model into one group,
14
+ leaving nothing to place. In practice it does not, and the reason is worth understanding because it
15
+ is the product's whole thesis. Take a typical application: ``User``, ``Order``, ``OrderLine``,
16
+ ``Product``, ``Event``. The first four are related and become one group. ``Event`` references
17
+ nothing and becomes its own. That is exactly the split that matters: the transactional core belongs
18
+ in a row store, the event stream belongs in a column store, and the entities nobody joins are
19
+ precisely the ones that were sitting in the wrong engine all along.
20
+
21
+ A client who wants two related entities in different engines can have that, and finds out about the
22
+ cost honestly: the relation stops being traversable, and the error at model-planning time says which
23
+ entities would have to share a group for the query to be possible.
24
+ """
25
+
26
+ from __future__ import annotations
27
+
28
+ from dataclasses import dataclass
29
+
30
+ from .model import LogicalModel
31
+
32
+ __all__ = ["Group", "colocation_groups", "group_of"]
33
+
34
+
35
+ @dataclass(frozen=True)
36
+ class Group:
37
+ """A set of entities placed together.
38
+
39
+ ``name`` is the alphabetically first member. It exists so that logs, proposals and error
40
+ messages can say ``group "order"`` instead of a hash, and it is only meaningful within one model
41
+ version - change the membership and you have changed the model, which changes its version.
42
+ """
43
+
44
+ name: str
45
+ members: tuple[str, ...]
46
+
47
+ def __contains__(self, entity: str) -> bool:
48
+ return entity in self.members
49
+
50
+
51
+ def colocation_groups(model: LogicalModel) -> tuple[Group, ...]:
52
+ """Connected components of the colocation graph, deterministically ordered.
53
+
54
+ Determinism here is not a nicety. The group name reaches the placement map, the telemetry and
55
+ the planner's decisions, so two runs over the same model have to produce the same names or the
56
+ control plane sees a model whose groups keep being renamed.
57
+ """
58
+ names = sorted(e.name for e in model.entities)
59
+ parent: dict[str, str] = {n: n for n in names}
60
+
61
+ def find(x: str) -> str:
62
+ while parent[x] != x:
63
+ parent[x] = parent[parent[x]]
64
+ x = parent[x]
65
+ return x
66
+
67
+ def union(a: str, b: str) -> None:
68
+ ra, rb = find(a), find(b)
69
+ if ra != rb:
70
+ # Always attach to the alphabetically smaller root, so the representative of a component
71
+ # does not depend on the order edges were visited in.
72
+ parent[max(ra, rb)] = min(ra, rb)
73
+
74
+ for relation in sorted(model.relations, key=lambda r: (r.source, r.name, r.target)):
75
+ union(relation.source, relation.target)
76
+ for atomic in model.atomic:
77
+ first = atomic[0]
78
+ for other in atomic[1:]:
79
+ union(first, other)
80
+
81
+ buckets: dict[str, list[str]] = {}
82
+ for name in names:
83
+ buckets.setdefault(find(name), []).append(name)
84
+
85
+ groups = [
86
+ Group(name=min(members), members=tuple(sorted(members)))
87
+ for members in buckets.values()
88
+ ]
89
+ groups.sort(key=lambda g: g.name)
90
+ return tuple(groups)
91
+
92
+
93
+ def group_of(groups: tuple[Group, ...], entity: str) -> Group:
94
+ for group in groups:
95
+ if entity in group:
96
+ return group
97
+ raise KeyError(f"{entity} is not in any group, which means it is not in the model")
sde/hashing.py ADDED
@@ -0,0 +1,242 @@
1
+ """Hiding identifier names from us, and from the model, without hiding them from the planner's job.
2
+
3
+ A client in a regulated environment may not want us to see that they have a table called
4
+ ``patient_diagnosis``. Requirement 11.3 lets them hash entity, field and relation names with a salt
5
+ that never leaves their infrastructure, and requirement 11.4 says the consequence must not be a
6
+ worse placement: no scoring feature may read the *meaning* of a name, so a hashed model has to be
7
+ placed identically to an unhashed one.
8
+
9
+ That second requirement is the interesting one. It is enforced by a test rather than by review, and
10
+ the test is worth more than a dozen unit tests of the planner, because it fails the moment anyone
11
+ writes ``if "log" in table_name`` anywhere in the decision path.
12
+
13
+ # Two consequences worth knowing before turning this on
14
+
15
+ **Hashing is a model change.** A group's name is its alphabetically first member, and hashing
16
+ changes which member that is; operation shape identifiers include the group and entity names, so
17
+ they change too. The canonical IR is different, so ``model_version`` is different, so the placement
18
+ map is different. Switching hashing on or off is therefore not a setting - it is a new model that
19
+ needs a new map, exactly as requirement 1.3 describes. Saying this plainly is cheaper than having a
20
+ client discover it when their existing map is refused.
21
+
22
+ **Your own tables get opaque names.** The physical layout comes from the map, which is keyed by
23
+ hashed names, so a table in the client's own database ends up called ``e_9c1f2a7b3d40``. For a
24
+ regulated deployment that is the point. For most deployments it is a real cost, and it should be
25
+ weighed rather than accepted by default - which is why this is off unless asked for.
26
+
27
+ # What is not hidden
28
+
29
+ Types, cardinalities, latencies, call counts, the shape of the graph. All of that is what the
30
+ planner actually reasons from, and none of it is a name. What the model loses is semantic signal:
31
+ ``orders`` tells a language model something that ``e_9c1f2a7b3d40`` does not, so proposals get
32
+ measurably worse (requirement 18.10). The deterministic path is unaffected, which is the whole point
33
+ of it not reading names.
34
+ """
35
+
36
+ from __future__ import annotations
37
+
38
+ import hashlib
39
+ import hmac
40
+ import os
41
+ import secrets
42
+ import unicodedata
43
+ from collections.abc import Mapping
44
+ from dataclasses import dataclass
45
+ from pathlib import Path
46
+
47
+ from .errors import DeclarationError
48
+ from .model import EntitySpec, FieldSpec, LogicalModel, RelationSpec, assemble
49
+
50
+ __all__ = ["NameMap", "hash_identifiers", "load_or_create_salt"]
51
+
52
+ # Twelve hex characters is 48 bits. Collisions matter here - two entities hashing to one name would
53
+ # merge them in the IR - so this is checked rather than assumed: hash_identifiers refuses on
54
+ # collision instead of producing a model with one entity where there were two.
55
+ DIGEST_CHARS = 12
56
+
57
+ _ENTITY_PREFIX = "e_"
58
+ _FIELD_PREFIX = "f_"
59
+ _RELATION_PREFIX = "r_"
60
+
61
+
62
+ def load_or_create_salt(path: Path | None = None) -> bytes:
63
+ """Read the client's salt, generating it on first use.
64
+
65
+ The salt never appears in a model, a telemetry window or anything sent to us: it is the whole
66
+ mechanism, and a salt we hold is a mechanism we could reverse. Stored with owner-only
67
+ permissions, because a readable salt in a shared container image is the same as no salt.
68
+
69
+ The file is read **verbatim**. An earlier version called ``.strip()`` on it, to be forgiving
70
+ about a trailing newline in a hand-made file, and that was a serious bug rather than a kindness.
71
+ ``bytes.strip()`` removes six byte *values* - space, tab, newline, carriage return, ``\x0b`` and
72
+ ``\x0c`` - and a random 32-byte salt begins or ends with one of them about 5% of the time.
73
+
74
+ When it did, the process that generated the salt used all 32 bytes and every process afterwards
75
+ used the stripped remainder. So one client computed **two different model versions** for one
76
+ declared model: the map issued for one is refused by the other, and a map that was accepted
77
+ names different tables. The TypeScript library takes the salt as bytes from its caller and
78
+ strips nothing, so a Python service and a Node service sharing one salt file disagreed too - the
79
+ exact failure the byte contract exists to prevent, arriving through the file rather than the
80
+ encoder.
81
+
82
+ A file created with ``echo`` therefore includes its newline, and the salt is those bytes
83
+ including it. That is consistent and checkable. Being forgiving would mean choosing an encoding,
84
+ hex or base64, and that changes what is on disk - which changes every name derived from it.
85
+ """
86
+ location = path or Path.home() / ".sde" / "salt"
87
+ if location.exists():
88
+ salt = location.read_bytes()
89
+ if len(salt) < 16:
90
+ raise DeclarationError(
91
+ f"the salt at {location} is shorter than 16 bytes. A short salt is "
92
+ "guessable, and a guessable salt means the names are not hidden. Delete the file "
93
+ "to have a new one generated - but note that a new salt is a new model version "
94
+ "and needs a new map."
95
+ )
96
+ return salt
97
+
98
+ location.parent.mkdir(parents=True, exist_ok=True)
99
+ salt = secrets.token_bytes(32)
100
+ # Written before the mode is set, then narrowed. os.open with 0o600 would be tighter, and is
101
+ # what this should become if the file ever holds more than a salt.
102
+ location.write_bytes(salt)
103
+ os.chmod(location, 0o600)
104
+ return salt
105
+
106
+
107
+ @dataclass(frozen=True)
108
+ class NameMap:
109
+ """The translation between what the client wrote and what we see.
110
+
111
+ Held only in the client's process. The library needs it because the placement map is keyed by
112
+ hashed names while the application still says ``session.save("User", ...)`` - so every lookup
113
+ crosses this boundary, and it is the only place that does.
114
+ """
115
+
116
+ entities: Mapping[str, str]
117
+ fields: Mapping[str, Mapping[str, str]]
118
+ relations: Mapping[str, Mapping[str, str]]
119
+
120
+ def entity(self, name: str) -> str:
121
+ try:
122
+ return self.entities[name]
123
+ except KeyError:
124
+ raise DeclarationError(
125
+ f"{name!r} is not in this model, so it has no hashed name. Either it was never "
126
+ "declared, or the model was rebuilt without it."
127
+ ) from None
128
+
129
+ def field(self, entity: str, name: str) -> str:
130
+ return self.fields[entity][name]
131
+
132
+ def relation(self, entity: str, name: str) -> str:
133
+ return self.relations[entity][name]
134
+
135
+
136
+ def _digest(salt: bytes, prefix: str, *parts: str) -> str:
137
+ """HMAC over the parts, NFC-normalised and joined by a separator no identifier can contain.
138
+
139
+ Two details, both of which a port can get wrong silently, so both are pinned in the format
140
+ contract.
141
+
142
+ **NFC first.** The canonical encoder normalises before it emits bytes, which is why two
143
+ libraries that declare ``Zamówienie`` in different normal forms compute the *same* model
144
+ version. Hashing a name before normalising it would throw that away: the same identifier written
145
+ two ways would give two digests, two model versions, and a placement map issued for one service
146
+ that the other refuses. Nothing about that is visible in ASCII, so it would have shipped and
147
+ then failed for a client whose entity names are not English.
148
+
149
+ **The separator, not concatenation.** Hashing ``("User", "id")`` as ``"Userid"`` would collide
150
+ with ``("Use", "rid")``. Unlikely is not a guarantee when the consequence is two fields becoming
151
+ one column. U+0000 cannot occur in an identifier, so the join is unambiguous.
152
+
153
+ The prefix is deliberately outside the HMAC. It labels the digest for a human reading a table
154
+ name; it carries no secret and adding it to the message would only make the derivation harder to
155
+ reproduce.
156
+ """
157
+ message = "\x00".join(unicodedata.normalize("NFC", part) for part in parts)
158
+ digest = hmac.new(salt, message.encode("utf-8"), hashlib.sha256).hexdigest()
159
+ return prefix + digest[:DIGEST_CHARS]
160
+
161
+
162
+ def hash_identifiers(model: LogicalModel, salt: bytes) -> tuple[LogicalModel, NameMap]:
163
+ """Return an equivalent model with every identifier replaced by a keyed digest.
164
+
165
+ Field and relation names are hashed *with their entity* in the input, so the same field name on
166
+ two entities produces two different digests. That is not paranoia: leaving them independent
167
+ would let us learn that two entities share a field called ``email`` even though we cannot read
168
+ the name, which is exactly the kind of structural leak hashing is meant to close.
169
+ """
170
+ if len(salt) < 16:
171
+ raise DeclarationError("the salt must be at least 16 bytes")
172
+
173
+ entity_names: dict[str, str] = {}
174
+ for spec in model.entities:
175
+ hashed = _digest(salt, _ENTITY_PREFIX, spec.name)
176
+ if hashed in entity_names.values():
177
+ clash = next(k for k, v in entity_names.items() if v == hashed)
178
+ raise DeclarationError(
179
+ f"{spec.name!r} and {clash!r} hash to the same name. Refused rather than merged: a "
180
+ "model with one entity where there were two would place both in one engine and "
181
+ "write both into one table. Change the salt."
182
+ )
183
+ entity_names[spec.name] = hashed
184
+
185
+ field_names: dict[str, dict[str, str]] = {}
186
+ relation_names: dict[str, dict[str, str]] = {}
187
+
188
+ entities: list[EntitySpec] = []
189
+ for spec in model.entities:
190
+ mapping: dict[str, str] = {}
191
+ for spec_field in spec.fields:
192
+ mapping[spec_field.name] = _digest(
193
+ salt, _FIELD_PREFIX, spec.name, spec_field.name
194
+ )
195
+ if len(set(mapping.values())) != len(mapping):
196
+ raise DeclarationError(
197
+ f"two fields of {spec.name!r} hash to the same name. Refused rather than merged. "
198
+ "Change the salt."
199
+ )
200
+ field_names[spec.name] = mapping
201
+
202
+ entities.append(
203
+ EntitySpec(
204
+ name=entity_names[spec.name],
205
+ fields=tuple(
206
+ FieldSpec(name=mapping[f.name], type=f.type, nullable=f.nullable)
207
+ for f in spec.fields
208
+ ),
209
+ key=tuple(mapping[k] for k in spec.key),
210
+ pii=tuple(mapping[p] for p in spec.pii),
211
+ # Residency is not an identifier. It is a jurisdiction, it is a hard constraint on
212
+ # placement, and hashing it would make the constraint unenforceable.
213
+ residency=spec.residency,
214
+ )
215
+ )
216
+
217
+ relations: list[RelationSpec] = []
218
+ for relation in model.relations:
219
+ hashed = _digest(salt, _RELATION_PREFIX, relation.source, relation.name)
220
+ relation_names.setdefault(relation.source, {})[relation.name] = hashed
221
+ relations.append(
222
+ RelationSpec(
223
+ name=hashed,
224
+ source=entity_names[relation.source],
225
+ target=entity_names[relation.target],
226
+ )
227
+ )
228
+
229
+ atomic = tuple(
230
+ tuple(sorted(entity_names[member] for member in group)) for group in model.atomic
231
+ )
232
+
233
+ hashed_model = assemble(
234
+ entities=tuple(entities),
235
+ relations=tuple(relations),
236
+ atomic=tuple(sorted(atomic)),
237
+ # The cost ceiling is a number and a currency, not an identifier.
238
+ cost_ceiling=model.cost_ceiling,
239
+ )
240
+ return hashed_model, NameMap(
241
+ entities=entity_names, fields=field_names, relations=relation_names
242
+ )