smart-data-engine-sdk 0.1.0__py3-none-any.whl
This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
- sde/__init__.py +318 -0
- sde/_cutover_project.py +179 -0
- sde/_local_state.py +188 -0
- sde/_operator_deadline.py +50 -0
- sde/_usage.py +314 -0
- sde/bulk.py +79 -0
- sde/canonical.py +141 -0
- sde/capabilities.py +62 -0
- sde/cutover.py +286 -0
- sde/engines/__init__.py +0 -0
- sde/engines/_clickhouse_connection.py +224 -0
- sde/engines/_index_build.py +294 -0
- sde/engines/_operator.py +394 -0
- sde/engines/_staging.py +222 -0
- sde/engines/_storage.py +22 -0
- sde/engines/_write_fences.py +271 -0
- sde/engines/clickhouse.py +1115 -0
- sde/engines/orderbook.py +457 -0
- sde/engines/postgres.py +967 -0
- sde/entity.py +170 -0
- sde/errors.py +103 -0
- sde/explain.py +300 -0
- sde/frozen_verification.py +152 -0
- sde/generation.py +131 -0
- sde/groups.py +97 -0
- sde/hashing.py +242 -0
- sde/index_build.py +313 -0
- sde/index_operator.py +347 -0
- sde/infer.py +461 -0
- sde/inspection.py +62 -0
- sde/internal.py +90 -0
- sde/layout.py +669 -0
- sde/local_cutover.py +801 -0
- sde/logging.py +143 -0
- sde/migration.py +856 -0
- sde/model.py +482 -0
- sde/physical.py +531 -0
- sde/placement.py +1010 -0
- sde/provisioning.py +63 -0
- sde/py.typed +0 -0
- sde/query.py +521 -0
- sde/routing.py +85 -0
- sde/schema.py +466 -0
- sde/session.py +993 -0
- sde/shapes.py +153 -0
- sde/staging.py +264 -0
- sde/staging_operator.py +393 -0
- sde/telemetry.py +1087 -0
- sde/testing/__init__.py +14 -0
- sde/testing/loader.py +175 -0
- sde/testing/memory.py +331 -0
- sde/types.py +228 -0
- sde/verification.py +220 -0
- sde/watermark.py +222 -0
- sde/write_fence.py +283 -0
- sde_demo/__init__.py +1 -0
- sde_demo/__main__.py +183 -0
- sde_demo/diagnostics.py +92 -0
- sde_demo/model.py +75 -0
- sde_demo/project.py +312 -0
- sde_demo/py.typed +0 -0
- sde_demo/query_count.py +301 -0
- sde_demo/resources.py +969 -0
- sde_demo/runtime.py +419 -0
- sde_demo/verification.py +242 -0
- sde_operator/__init__.py +1 -0
- sde_operator/__main__.py +210 -0
- smart_data_engine_sdk-0.1.0.dist-info/METADATA +174 -0
- smart_data_engine_sdk-0.1.0.dist-info/RECORD +73 -0
- smart_data_engine_sdk-0.1.0.dist-info/WHEEL +4 -0
- smart_data_engine_sdk-0.1.0.dist-info/entry_points.txt +3 -0
- smart_data_engine_sdk-0.1.0.dist-info/licenses/LICENSE +201 -0
- smart_data_engine_sdk-0.1.0.dist-info/licenses/NOTICE +13 -0
|
@@ -0,0 +1,152 @@
|
|
|
1
|
+
"""Drain named native barriers and compare stable copies, including target-only rows.
|
|
2
|
+
|
|
3
|
+
This is a client-side data operation. It neither chooses a placement nor activates a map. It
|
|
4
|
+
leaves its barriers installed on success and failure; the executor owns the durable decision
|
|
5
|
+
which permits release, rollback or activation.
|
|
6
|
+
"""
|
|
7
|
+
|
|
8
|
+
from __future__ import annotations
|
|
9
|
+
|
|
10
|
+
import re
|
|
11
|
+
from collections.abc import Mapping
|
|
12
|
+
from dataclasses import dataclass
|
|
13
|
+
from time import perf_counter_ns
|
|
14
|
+
from typing import Any, cast
|
|
15
|
+
|
|
16
|
+
from .errors import MigrationRefused
|
|
17
|
+
from .generation import Fencable, check_epoch
|
|
18
|
+
from .inspection import InspectionContext
|
|
19
|
+
from .migration import CHUNK_ROWS, VerifyReport, verify
|
|
20
|
+
from .verification import VerificationRequest
|
|
21
|
+
from .write_fence import FenceState, WriteFence
|
|
22
|
+
|
|
23
|
+
|
|
24
|
+
@dataclass(frozen=True)
|
|
25
|
+
class FrozenTable:
|
|
26
|
+
engine: str
|
|
27
|
+
materialization: str
|
|
28
|
+
table: str
|
|
29
|
+
identity: str
|
|
30
|
+
project_id: str
|
|
31
|
+
epoch: int
|
|
32
|
+
hold_id: str
|
|
33
|
+
|
|
34
|
+
def as_record(self) -> dict[str, Any]:
|
|
35
|
+
return {
|
|
36
|
+
"engine": self.engine,
|
|
37
|
+
"materialization": self.materialization,
|
|
38
|
+
"table": self.table,
|
|
39
|
+
"identity": self.identity,
|
|
40
|
+
"project_id": self.project_id,
|
|
41
|
+
"epoch": self.epoch,
|
|
42
|
+
"hold_id": self.hold_id,
|
|
43
|
+
}
|
|
44
|
+
|
|
45
|
+
|
|
46
|
+
@dataclass(frozen=True)
|
|
47
|
+
class FrozenVerifyReport:
|
|
48
|
+
comparison: VerifyReport
|
|
49
|
+
barriers: tuple[FrozenTable, ...]
|
|
50
|
+
elapsed_ms: int
|
|
51
|
+
|
|
52
|
+
@property
|
|
53
|
+
def matched(self) -> bool:
|
|
54
|
+
return (
|
|
55
|
+
self.comparison.matched and self.comparison.rows_source == self.comparison.rows_target
|
|
56
|
+
)
|
|
57
|
+
|
|
58
|
+
def as_record(self) -> dict[str, Any]:
|
|
59
|
+
return {
|
|
60
|
+
"protocol": 1,
|
|
61
|
+
"comparison": self.comparison.as_record(),
|
|
62
|
+
"barriers": [barrier.as_record() for barrier in self.barriers],
|
|
63
|
+
"elapsed_ms": self.elapsed_ms,
|
|
64
|
+
"matched": self.matched,
|
|
65
|
+
}
|
|
66
|
+
|
|
67
|
+
|
|
68
|
+
def _matches(state: FenceState, wanted: FrozenTable) -> None:
|
|
69
|
+
if state.identity != wanted.identity or state.project_id != wanted.project_id:
|
|
70
|
+
raise MigrationRefused("a frozen comparison table changed identity or project")
|
|
71
|
+
if not state.complete or state.epoch != wanted.epoch or wanted.hold_id not in state.holds:
|
|
72
|
+
raise MigrationRefused("a frozen comparison lost its named barrier or write generation")
|
|
73
|
+
|
|
74
|
+
|
|
75
|
+
def verify_frozen(
|
|
76
|
+
context: InspectionContext,
|
|
77
|
+
group: str,
|
|
78
|
+
*,
|
|
79
|
+
request: VerificationRequest,
|
|
80
|
+
hold_id: str,
|
|
81
|
+
epochs: Mapping[str, int],
|
|
82
|
+
chunk_rows: int = CHUNK_ROWS,
|
|
83
|
+
at: str | None = None,
|
|
84
|
+
) -> FrozenVerifyReport:
|
|
85
|
+
"""Acquire/drain each barrier and compare exact logical content; do not release the holds.
|
|
86
|
+
|
|
87
|
+
`epochs` names materialization ids, including the source. They can differ during operator
|
|
88
|
+
maintenance and therefore cannot be inferred from a runtime session's single group epoch.
|
|
89
|
+
A caller must budget this work and handle uncertain engine I/O as recovery, not as success.
|
|
90
|
+
"""
|
|
91
|
+
if not isinstance(hold_id, str) or re.fullmatch("[0-9a-f]{32}", hold_id) is None:
|
|
92
|
+
raise MigrationRefused(
|
|
93
|
+
"a frozen comparison hold id must be 32 lowercase hexadecimal digits"
|
|
94
|
+
)
|
|
95
|
+
if chunk_rows < 1:
|
|
96
|
+
raise MigrationRefused("a frozen comparison needs a positive chunk size")
|
|
97
|
+
request.check_session(context.placement, project_id=context.project_id, group=group)
|
|
98
|
+
if at is not None:
|
|
99
|
+
request.check_time(at)
|
|
100
|
+
spot = context.placement.placement_of(group)
|
|
101
|
+
materials = (spot.source, *spot.also_write)
|
|
102
|
+
if not spot.also_write or set(epochs) != {material.id for material in materials}:
|
|
103
|
+
raise MigrationRefused("frozen comparison epochs must name exactly the source and copy ids")
|
|
104
|
+
checked_epochs = {name: check_epoch(value) for name, value in epochs.items()}
|
|
105
|
+
planned: list[tuple[WriteFence, FrozenTable]] = []
|
|
106
|
+
for material in sorted(materials, key=lambda value: (value.engine, value.id)):
|
|
107
|
+
engine = context.engines[material.engine]
|
|
108
|
+
if not callable(getattr(engine, "write_fence", None)) or not callable(
|
|
109
|
+
getattr(engine, "validate_schema", None)
|
|
110
|
+
):
|
|
111
|
+
raise MigrationRefused(
|
|
112
|
+
"frozen comparison requires native write fences and schema checks"
|
|
113
|
+
)
|
|
114
|
+
native = cast(Fencable, engine)
|
|
115
|
+
native.validate_schema(material.layout)
|
|
116
|
+
for table in sorted(material.layout.tables.values()):
|
|
117
|
+
fence = native.write_fence(table, project_id=context.project_id)
|
|
118
|
+
state = fence.state()
|
|
119
|
+
epoch = checked_epochs[material.id]
|
|
120
|
+
if hold_id in state.retired:
|
|
121
|
+
raise MigrationRefused("a frozen comparison cannot reuse a retired barrier id")
|
|
122
|
+
if not state.complete or state.epoch != epoch:
|
|
123
|
+
raise MigrationRefused(
|
|
124
|
+
"a frozen comparison table is not at the expected write generation"
|
|
125
|
+
)
|
|
126
|
+
planned.append(
|
|
127
|
+
(
|
|
128
|
+
fence,
|
|
129
|
+
FrozenTable(
|
|
130
|
+
material.engine,
|
|
131
|
+
material.id,
|
|
132
|
+
table,
|
|
133
|
+
state.identity,
|
|
134
|
+
context.project_id,
|
|
135
|
+
epoch,
|
|
136
|
+
hold_id,
|
|
137
|
+
),
|
|
138
|
+
)
|
|
139
|
+
)
|
|
140
|
+
started = perf_counter_ns()
|
|
141
|
+
for fence, wanted in planned:
|
|
142
|
+
# Constraint presence alone does not drain an old ClickHouse INSERT. Always execute the
|
|
143
|
+
# native drain, including when resuming an already installed hold.
|
|
144
|
+
_matches(fence.freeze(hold_id), wanted)
|
|
145
|
+
comparison = verify(context, group, request=request, chunk_rows=chunk_rows, at=at)
|
|
146
|
+
for fence, wanted in planned:
|
|
147
|
+
_matches(fence.state(), wanted)
|
|
148
|
+
return FrozenVerifyReport(
|
|
149
|
+
comparison,
|
|
150
|
+
tuple(wanted for _, wanted in planned),
|
|
151
|
+
(perf_counter_ns() - started) // 1_000_000,
|
|
152
|
+
)
|
sde/generation.py
ADDED
|
@@ -0,0 +1,131 @@
|
|
|
1
|
+
"""Wire vocabulary for generation-bearing maps, independent of drivers and placement classes."""
|
|
2
|
+
|
|
3
|
+
from __future__ import annotations
|
|
4
|
+
|
|
5
|
+
from collections.abc import Mapping, Sequence
|
|
6
|
+
from typing import TYPE_CHECKING, Any, Protocol, cast
|
|
7
|
+
|
|
8
|
+
from .errors import MigrationRefused
|
|
9
|
+
|
|
10
|
+
if TYPE_CHECKING:
|
|
11
|
+
from .model import LogicalModel
|
|
12
|
+
from .physical import PhysicalFinding
|
|
13
|
+
from .placement import PhysicalLayout, PlacementMap
|
|
14
|
+
from .session import Engine
|
|
15
|
+
from .write_fence import WriteFence
|
|
16
|
+
|
|
17
|
+
|
|
18
|
+
class Fencable(Protocol):
|
|
19
|
+
def write_fence(self, table: str, *, project_id: str) -> WriteFence: ...
|
|
20
|
+
def validate_schema(
|
|
21
|
+
self, layout: PhysicalLayout, *, keys: Mapping[str, Sequence[str]] | None = None
|
|
22
|
+
) -> tuple[PhysicalFinding, ...]: ...
|
|
23
|
+
|
|
24
|
+
|
|
25
|
+
DRAIN_TABLE = "__sde_fence_drains"
|
|
26
|
+
EPOCH_COLUMN = "__sde_write_epoch"
|
|
27
|
+
MAX_EPOCH = 9_007_199_254_740_991
|
|
28
|
+
GENERATIONS_SINCE = 4
|
|
29
|
+
"""The placement map contract that introduced ``project_id`` and per-group ``write_epoch``."""
|
|
30
|
+
|
|
31
|
+
|
|
32
|
+
def check_epoch(epoch: int) -> int:
|
|
33
|
+
if type(epoch) not in (int, float) or not 1 <= epoch <= MAX_EPOCH or int(epoch) != epoch:
|
|
34
|
+
raise MigrationRefused("write epoch must be a positive safe integer")
|
|
35
|
+
return int(epoch)
|
|
36
|
+
|
|
37
|
+
|
|
38
|
+
def json_numbers(value: Any) -> Any:
|
|
39
|
+
"""JSON has one number type. Match integral JS numbers before checking/signing a v4 map.
|
|
40
|
+
|
|
41
|
+
Canonical encoding itself remains strict. Nonintegral/nonfinite numbers survive this pass and
|
|
42
|
+
are refused by the map's canonical boundary. Earlier map contracts retain their old behavior.
|
|
43
|
+
"""
|
|
44
|
+
if isinstance(value, float) and value.is_integer():
|
|
45
|
+
return int(value)
|
|
46
|
+
if isinstance(value, dict):
|
|
47
|
+
return {key: json_numbers(item) for key, item in value.items()}
|
|
48
|
+
if isinstance(value, list):
|
|
49
|
+
return [json_numbers(item) for item in value]
|
|
50
|
+
if isinstance(value, tuple):
|
|
51
|
+
return tuple(json_numbers(item) for item in value)
|
|
52
|
+
return value
|
|
53
|
+
|
|
54
|
+
|
|
55
|
+
def check_map_project(placement: PlacementMap, project_id: str | None) -> str | None:
|
|
56
|
+
if placement.contract < 4:
|
|
57
|
+
return None
|
|
58
|
+
if project_id is None or project_id != placement.project_id:
|
|
59
|
+
raise MigrationRefused(
|
|
60
|
+
"this generation-bearing map needs its locally configured project_id; "
|
|
61
|
+
"do not learn that identity from the supplied map"
|
|
62
|
+
)
|
|
63
|
+
if placement.fingerprint is None:
|
|
64
|
+
raise MigrationRefused("generation-bearing sessions need an immutable loaded placement map")
|
|
65
|
+
return project_id
|
|
66
|
+
|
|
67
|
+
|
|
68
|
+
def validate_generations(
|
|
69
|
+
model: LogicalModel,
|
|
70
|
+
placement: PlacementMap,
|
|
71
|
+
engines: Mapping[str, Engine],
|
|
72
|
+
project_id: str | None,
|
|
73
|
+
) -> tuple[PhysicalFinding, ...]:
|
|
74
|
+
"""Check generations and columns; return how tables differ from the declared physical design.
|
|
75
|
+
|
|
76
|
+
The physical design is reported rather than refused: a running application must not stop
|
|
77
|
+
because a table's sort key or index differs from the map (requirement 3.6). Columns, types
|
|
78
|
+
and generations still refuse.
|
|
79
|
+
"""
|
|
80
|
+
if placement.contract < 4:
|
|
81
|
+
return ()
|
|
82
|
+
findings: list[PhysicalFinding] = []
|
|
83
|
+
local_project = check_map_project(placement, project_id)
|
|
84
|
+
assert local_project is not None
|
|
85
|
+
if model.version != placement.model_version:
|
|
86
|
+
raise MigrationRefused("the generation-bearing map names another session model")
|
|
87
|
+
for name in sorted(placement.groups):
|
|
88
|
+
spot = placement.groups[name]
|
|
89
|
+
for material in spot.all():
|
|
90
|
+
factory = getattr(engines[material.engine], "write_fence", None)
|
|
91
|
+
if not callable(factory) or not callable(
|
|
92
|
+
getattr(engines[material.engine], "validate_schema", None)
|
|
93
|
+
):
|
|
94
|
+
raise MigrationRefused(
|
|
95
|
+
f"engine {material.engine} does not implement write generations"
|
|
96
|
+
)
|
|
97
|
+
keys = {entity: model.entity(entity).key for entity in material.layout.tables}
|
|
98
|
+
findings.extend(
|
|
99
|
+
cast(Fencable, engines[material.engine]).validate_schema(
|
|
100
|
+
material.layout, keys=keys
|
|
101
|
+
)
|
|
102
|
+
)
|
|
103
|
+
for table in sorted(material.layout.tables.values()):
|
|
104
|
+
state = (
|
|
105
|
+
cast(Fencable, engines[material.engine])
|
|
106
|
+
.write_fence(table, project_id=local_project)
|
|
107
|
+
.state()
|
|
108
|
+
)
|
|
109
|
+
if not state.complete or state.epoch != spot.write_epoch:
|
|
110
|
+
raise MigrationRefused(
|
|
111
|
+
f"the write generation for {name} is not active in {material.engine}; "
|
|
112
|
+
"provision the signed map or load the current map before opening a session"
|
|
113
|
+
)
|
|
114
|
+
return tuple(findings)
|
|
115
|
+
|
|
116
|
+
|
|
117
|
+
def stamp_values(
|
|
118
|
+
placement: PlacementMap, group: str, values: Mapping[str, Any]
|
|
119
|
+
) -> Mapping[str, Any]:
|
|
120
|
+
epoch = placement.placement_of(group).write_epoch
|
|
121
|
+
if epoch is None:
|
|
122
|
+
return values
|
|
123
|
+
if EPOCH_COLUMN in values:
|
|
124
|
+
raise MigrationRefused("the write-epoch column is reserved for the SDK")
|
|
125
|
+
return {**values, EPOCH_COLUMN: epoch}
|
|
126
|
+
|
|
127
|
+
|
|
128
|
+
def logical_row(placement: PlacementMap, row: Any) -> Any:
|
|
129
|
+
if placement.contract >= 4 and isinstance(row, dict):
|
|
130
|
+
return {key: value for key, value in row.items() if key != EPOCH_COLUMN}
|
|
131
|
+
return row
|
sde/groups.py
ADDED
|
@@ -0,0 +1,97 @@
|
|
|
1
|
+
"""Colocation groups: the unit of placement.
|
|
2
|
+
|
|
3
|
+
Entities that are queried together, or that must change together, live in the same engine. The graph
|
|
4
|
+
has an edge for every relation and for every declared atomicity, and a group is a connected
|
|
5
|
+
component of it.
|
|
6
|
+
|
|
7
|
+
This looks like a limitation and is the opposite. A join across two engines means pulling both sides
|
|
8
|
+
over the network and joining in the client's process: slow, memory-hungry, and the single easiest
|
|
9
|
+
way for a young product to embarrass itself. Making colocation a constraint turns that problem into
|
|
10
|
+
something the planner simply respects, and the consistency contract falls straight out of it - one
|
|
11
|
+
group, one engine, that engine's transaction semantics, and no distributed transactions anywhere.
|
|
12
|
+
|
|
13
|
+
The obvious worry is that every relation being an edge collapses a normalised model into one group,
|
|
14
|
+
leaving nothing to place. In practice it does not, and the reason is worth understanding because it
|
|
15
|
+
is the product's whole thesis. Take a typical application: ``User``, ``Order``, ``OrderLine``,
|
|
16
|
+
``Product``, ``Event``. The first four are related and become one group. ``Event`` references
|
|
17
|
+
nothing and becomes its own. That is exactly the split that matters: the transactional core belongs
|
|
18
|
+
in a row store, the event stream belongs in a column store, and the entities nobody joins are
|
|
19
|
+
precisely the ones that were sitting in the wrong engine all along.
|
|
20
|
+
|
|
21
|
+
A client who wants two related entities in different engines can have that, and finds out about the
|
|
22
|
+
cost honestly: the relation stops being traversable, and the error at model-planning time says which
|
|
23
|
+
entities would have to share a group for the query to be possible.
|
|
24
|
+
"""
|
|
25
|
+
|
|
26
|
+
from __future__ import annotations
|
|
27
|
+
|
|
28
|
+
from dataclasses import dataclass
|
|
29
|
+
|
|
30
|
+
from .model import LogicalModel
|
|
31
|
+
|
|
32
|
+
__all__ = ["Group", "colocation_groups", "group_of"]
|
|
33
|
+
|
|
34
|
+
|
|
35
|
+
@dataclass(frozen=True)
|
|
36
|
+
class Group:
|
|
37
|
+
"""A set of entities placed together.
|
|
38
|
+
|
|
39
|
+
``name`` is the alphabetically first member. It exists so that logs, proposals and error
|
|
40
|
+
messages can say ``group "order"`` instead of a hash, and it is only meaningful within one model
|
|
41
|
+
version - change the membership and you have changed the model, which changes its version.
|
|
42
|
+
"""
|
|
43
|
+
|
|
44
|
+
name: str
|
|
45
|
+
members: tuple[str, ...]
|
|
46
|
+
|
|
47
|
+
def __contains__(self, entity: str) -> bool:
|
|
48
|
+
return entity in self.members
|
|
49
|
+
|
|
50
|
+
|
|
51
|
+
def colocation_groups(model: LogicalModel) -> tuple[Group, ...]:
|
|
52
|
+
"""Connected components of the colocation graph, deterministically ordered.
|
|
53
|
+
|
|
54
|
+
Determinism here is not a nicety. The group name reaches the placement map, the telemetry and
|
|
55
|
+
the planner's decisions, so two runs over the same model have to produce the same names or the
|
|
56
|
+
control plane sees a model whose groups keep being renamed.
|
|
57
|
+
"""
|
|
58
|
+
names = sorted(e.name for e in model.entities)
|
|
59
|
+
parent: dict[str, str] = {n: n for n in names}
|
|
60
|
+
|
|
61
|
+
def find(x: str) -> str:
|
|
62
|
+
while parent[x] != x:
|
|
63
|
+
parent[x] = parent[parent[x]]
|
|
64
|
+
x = parent[x]
|
|
65
|
+
return x
|
|
66
|
+
|
|
67
|
+
def union(a: str, b: str) -> None:
|
|
68
|
+
ra, rb = find(a), find(b)
|
|
69
|
+
if ra != rb:
|
|
70
|
+
# Always attach to the alphabetically smaller root, so the representative of a component
|
|
71
|
+
# does not depend on the order edges were visited in.
|
|
72
|
+
parent[max(ra, rb)] = min(ra, rb)
|
|
73
|
+
|
|
74
|
+
for relation in sorted(model.relations, key=lambda r: (r.source, r.name, r.target)):
|
|
75
|
+
union(relation.source, relation.target)
|
|
76
|
+
for atomic in model.atomic:
|
|
77
|
+
first = atomic[0]
|
|
78
|
+
for other in atomic[1:]:
|
|
79
|
+
union(first, other)
|
|
80
|
+
|
|
81
|
+
buckets: dict[str, list[str]] = {}
|
|
82
|
+
for name in names:
|
|
83
|
+
buckets.setdefault(find(name), []).append(name)
|
|
84
|
+
|
|
85
|
+
groups = [
|
|
86
|
+
Group(name=min(members), members=tuple(sorted(members)))
|
|
87
|
+
for members in buckets.values()
|
|
88
|
+
]
|
|
89
|
+
groups.sort(key=lambda g: g.name)
|
|
90
|
+
return tuple(groups)
|
|
91
|
+
|
|
92
|
+
|
|
93
|
+
def group_of(groups: tuple[Group, ...], entity: str) -> Group:
|
|
94
|
+
for group in groups:
|
|
95
|
+
if entity in group:
|
|
96
|
+
return group
|
|
97
|
+
raise KeyError(f"{entity} is not in any group, which means it is not in the model")
|
sde/hashing.py
ADDED
|
@@ -0,0 +1,242 @@
|
|
|
1
|
+
"""Hiding identifier names from us, and from the model, without hiding them from the planner's job.
|
|
2
|
+
|
|
3
|
+
A client in a regulated environment may not want us to see that they have a table called
|
|
4
|
+
``patient_diagnosis``. Requirement 11.3 lets them hash entity, field and relation names with a salt
|
|
5
|
+
that never leaves their infrastructure, and requirement 11.4 says the consequence must not be a
|
|
6
|
+
worse placement: no scoring feature may read the *meaning* of a name, so a hashed model has to be
|
|
7
|
+
placed identically to an unhashed one.
|
|
8
|
+
|
|
9
|
+
That second requirement is the interesting one. It is enforced by a test rather than by review, and
|
|
10
|
+
the test is worth more than a dozen unit tests of the planner, because it fails the moment anyone
|
|
11
|
+
writes ``if "log" in table_name`` anywhere in the decision path.
|
|
12
|
+
|
|
13
|
+
# Two consequences worth knowing before turning this on
|
|
14
|
+
|
|
15
|
+
**Hashing is a model change.** A group's name is its alphabetically first member, and hashing
|
|
16
|
+
changes which member that is; operation shape identifiers include the group and entity names, so
|
|
17
|
+
they change too. The canonical IR is different, so ``model_version`` is different, so the placement
|
|
18
|
+
map is different. Switching hashing on or off is therefore not a setting - it is a new model that
|
|
19
|
+
needs a new map, exactly as requirement 1.3 describes. Saying this plainly is cheaper than having a
|
|
20
|
+
client discover it when their existing map is refused.
|
|
21
|
+
|
|
22
|
+
**Your own tables get opaque names.** The physical layout comes from the map, which is keyed by
|
|
23
|
+
hashed names, so a table in the client's own database ends up called ``e_9c1f2a7b3d40``. For a
|
|
24
|
+
regulated deployment that is the point. For most deployments it is a real cost, and it should be
|
|
25
|
+
weighed rather than accepted by default - which is why this is off unless asked for.
|
|
26
|
+
|
|
27
|
+
# What is not hidden
|
|
28
|
+
|
|
29
|
+
Types, cardinalities, latencies, call counts, the shape of the graph. All of that is what the
|
|
30
|
+
planner actually reasons from, and none of it is a name. What the model loses is semantic signal:
|
|
31
|
+
``orders`` tells a language model something that ``e_9c1f2a7b3d40`` does not, so proposals get
|
|
32
|
+
measurably worse (requirement 18.10). The deterministic path is unaffected, which is the whole point
|
|
33
|
+
of it not reading names.
|
|
34
|
+
"""
|
|
35
|
+
|
|
36
|
+
from __future__ import annotations
|
|
37
|
+
|
|
38
|
+
import hashlib
|
|
39
|
+
import hmac
|
|
40
|
+
import os
|
|
41
|
+
import secrets
|
|
42
|
+
import unicodedata
|
|
43
|
+
from collections.abc import Mapping
|
|
44
|
+
from dataclasses import dataclass
|
|
45
|
+
from pathlib import Path
|
|
46
|
+
|
|
47
|
+
from .errors import DeclarationError
|
|
48
|
+
from .model import EntitySpec, FieldSpec, LogicalModel, RelationSpec, assemble
|
|
49
|
+
|
|
50
|
+
__all__ = ["NameMap", "hash_identifiers", "load_or_create_salt"]
|
|
51
|
+
|
|
52
|
+
# Twelve hex characters is 48 bits. Collisions matter here - two entities hashing to one name would
|
|
53
|
+
# merge them in the IR - so this is checked rather than assumed: hash_identifiers refuses on
|
|
54
|
+
# collision instead of producing a model with one entity where there were two.
|
|
55
|
+
DIGEST_CHARS = 12
|
|
56
|
+
|
|
57
|
+
_ENTITY_PREFIX = "e_"
|
|
58
|
+
_FIELD_PREFIX = "f_"
|
|
59
|
+
_RELATION_PREFIX = "r_"
|
|
60
|
+
|
|
61
|
+
|
|
62
|
+
def load_or_create_salt(path: Path | None = None) -> bytes:
|
|
63
|
+
"""Read the client's salt, generating it on first use.
|
|
64
|
+
|
|
65
|
+
The salt never appears in a model, a telemetry window or anything sent to us: it is the whole
|
|
66
|
+
mechanism, and a salt we hold is a mechanism we could reverse. Stored with owner-only
|
|
67
|
+
permissions, because a readable salt in a shared container image is the same as no salt.
|
|
68
|
+
|
|
69
|
+
The file is read **verbatim**. An earlier version called ``.strip()`` on it, to be forgiving
|
|
70
|
+
about a trailing newline in a hand-made file, and that was a serious bug rather than a kindness.
|
|
71
|
+
``bytes.strip()`` removes six byte *values* - space, tab, newline, carriage return, ``\x0b`` and
|
|
72
|
+
``\x0c`` - and a random 32-byte salt begins or ends with one of them about 5% of the time.
|
|
73
|
+
|
|
74
|
+
When it did, the process that generated the salt used all 32 bytes and every process afterwards
|
|
75
|
+
used the stripped remainder. So one client computed **two different model versions** for one
|
|
76
|
+
declared model: the map issued for one is refused by the other, and a map that was accepted
|
|
77
|
+
names different tables. The TypeScript library takes the salt as bytes from its caller and
|
|
78
|
+
strips nothing, so a Python service and a Node service sharing one salt file disagreed too - the
|
|
79
|
+
exact failure the byte contract exists to prevent, arriving through the file rather than the
|
|
80
|
+
encoder.
|
|
81
|
+
|
|
82
|
+
A file created with ``echo`` therefore includes its newline, and the salt is those bytes
|
|
83
|
+
including it. That is consistent and checkable. Being forgiving would mean choosing an encoding,
|
|
84
|
+
hex or base64, and that changes what is on disk - which changes every name derived from it.
|
|
85
|
+
"""
|
|
86
|
+
location = path or Path.home() / ".sde" / "salt"
|
|
87
|
+
if location.exists():
|
|
88
|
+
salt = location.read_bytes()
|
|
89
|
+
if len(salt) < 16:
|
|
90
|
+
raise DeclarationError(
|
|
91
|
+
f"the salt at {location} is shorter than 16 bytes. A short salt is "
|
|
92
|
+
"guessable, and a guessable salt means the names are not hidden. Delete the file "
|
|
93
|
+
"to have a new one generated - but note that a new salt is a new model version "
|
|
94
|
+
"and needs a new map."
|
|
95
|
+
)
|
|
96
|
+
return salt
|
|
97
|
+
|
|
98
|
+
location.parent.mkdir(parents=True, exist_ok=True)
|
|
99
|
+
salt = secrets.token_bytes(32)
|
|
100
|
+
# Written before the mode is set, then narrowed. os.open with 0o600 would be tighter, and is
|
|
101
|
+
# what this should become if the file ever holds more than a salt.
|
|
102
|
+
location.write_bytes(salt)
|
|
103
|
+
os.chmod(location, 0o600)
|
|
104
|
+
return salt
|
|
105
|
+
|
|
106
|
+
|
|
107
|
+
@dataclass(frozen=True)
|
|
108
|
+
class NameMap:
|
|
109
|
+
"""The translation between what the client wrote and what we see.
|
|
110
|
+
|
|
111
|
+
Held only in the client's process. The library needs it because the placement map is keyed by
|
|
112
|
+
hashed names while the application still says ``session.save("User", ...)`` - so every lookup
|
|
113
|
+
crosses this boundary, and it is the only place that does.
|
|
114
|
+
"""
|
|
115
|
+
|
|
116
|
+
entities: Mapping[str, str]
|
|
117
|
+
fields: Mapping[str, Mapping[str, str]]
|
|
118
|
+
relations: Mapping[str, Mapping[str, str]]
|
|
119
|
+
|
|
120
|
+
def entity(self, name: str) -> str:
|
|
121
|
+
try:
|
|
122
|
+
return self.entities[name]
|
|
123
|
+
except KeyError:
|
|
124
|
+
raise DeclarationError(
|
|
125
|
+
f"{name!r} is not in this model, so it has no hashed name. Either it was never "
|
|
126
|
+
"declared, or the model was rebuilt without it."
|
|
127
|
+
) from None
|
|
128
|
+
|
|
129
|
+
def field(self, entity: str, name: str) -> str:
|
|
130
|
+
return self.fields[entity][name]
|
|
131
|
+
|
|
132
|
+
def relation(self, entity: str, name: str) -> str:
|
|
133
|
+
return self.relations[entity][name]
|
|
134
|
+
|
|
135
|
+
|
|
136
|
+
def _digest(salt: bytes, prefix: str, *parts: str) -> str:
|
|
137
|
+
"""HMAC over the parts, NFC-normalised and joined by a separator no identifier can contain.
|
|
138
|
+
|
|
139
|
+
Two details, both of which a port can get wrong silently, so both are pinned in the format
|
|
140
|
+
contract.
|
|
141
|
+
|
|
142
|
+
**NFC first.** The canonical encoder normalises before it emits bytes, which is why two
|
|
143
|
+
libraries that declare ``Zamówienie`` in different normal forms compute the *same* model
|
|
144
|
+
version. Hashing a name before normalising it would throw that away: the same identifier written
|
|
145
|
+
two ways would give two digests, two model versions, and a placement map issued for one service
|
|
146
|
+
that the other refuses. Nothing about that is visible in ASCII, so it would have shipped and
|
|
147
|
+
then failed for a client whose entity names are not English.
|
|
148
|
+
|
|
149
|
+
**The separator, not concatenation.** Hashing ``("User", "id")`` as ``"Userid"`` would collide
|
|
150
|
+
with ``("Use", "rid")``. Unlikely is not a guarantee when the consequence is two fields becoming
|
|
151
|
+
one column. U+0000 cannot occur in an identifier, so the join is unambiguous.
|
|
152
|
+
|
|
153
|
+
The prefix is deliberately outside the HMAC. It labels the digest for a human reading a table
|
|
154
|
+
name; it carries no secret and adding it to the message would only make the derivation harder to
|
|
155
|
+
reproduce.
|
|
156
|
+
"""
|
|
157
|
+
message = "\x00".join(unicodedata.normalize("NFC", part) for part in parts)
|
|
158
|
+
digest = hmac.new(salt, message.encode("utf-8"), hashlib.sha256).hexdigest()
|
|
159
|
+
return prefix + digest[:DIGEST_CHARS]
|
|
160
|
+
|
|
161
|
+
|
|
162
|
+
def hash_identifiers(model: LogicalModel, salt: bytes) -> tuple[LogicalModel, NameMap]:
|
|
163
|
+
"""Return an equivalent model with every identifier replaced by a keyed digest.
|
|
164
|
+
|
|
165
|
+
Field and relation names are hashed *with their entity* in the input, so the same field name on
|
|
166
|
+
two entities produces two different digests. That is not paranoia: leaving them independent
|
|
167
|
+
would let us learn that two entities share a field called ``email`` even though we cannot read
|
|
168
|
+
the name, which is exactly the kind of structural leak hashing is meant to close.
|
|
169
|
+
"""
|
|
170
|
+
if len(salt) < 16:
|
|
171
|
+
raise DeclarationError("the salt must be at least 16 bytes")
|
|
172
|
+
|
|
173
|
+
entity_names: dict[str, str] = {}
|
|
174
|
+
for spec in model.entities:
|
|
175
|
+
hashed = _digest(salt, _ENTITY_PREFIX, spec.name)
|
|
176
|
+
if hashed in entity_names.values():
|
|
177
|
+
clash = next(k for k, v in entity_names.items() if v == hashed)
|
|
178
|
+
raise DeclarationError(
|
|
179
|
+
f"{spec.name!r} and {clash!r} hash to the same name. Refused rather than merged: a "
|
|
180
|
+
"model with one entity where there were two would place both in one engine and "
|
|
181
|
+
"write both into one table. Change the salt."
|
|
182
|
+
)
|
|
183
|
+
entity_names[spec.name] = hashed
|
|
184
|
+
|
|
185
|
+
field_names: dict[str, dict[str, str]] = {}
|
|
186
|
+
relation_names: dict[str, dict[str, str]] = {}
|
|
187
|
+
|
|
188
|
+
entities: list[EntitySpec] = []
|
|
189
|
+
for spec in model.entities:
|
|
190
|
+
mapping: dict[str, str] = {}
|
|
191
|
+
for spec_field in spec.fields:
|
|
192
|
+
mapping[spec_field.name] = _digest(
|
|
193
|
+
salt, _FIELD_PREFIX, spec.name, spec_field.name
|
|
194
|
+
)
|
|
195
|
+
if len(set(mapping.values())) != len(mapping):
|
|
196
|
+
raise DeclarationError(
|
|
197
|
+
f"two fields of {spec.name!r} hash to the same name. Refused rather than merged. "
|
|
198
|
+
"Change the salt."
|
|
199
|
+
)
|
|
200
|
+
field_names[spec.name] = mapping
|
|
201
|
+
|
|
202
|
+
entities.append(
|
|
203
|
+
EntitySpec(
|
|
204
|
+
name=entity_names[spec.name],
|
|
205
|
+
fields=tuple(
|
|
206
|
+
FieldSpec(name=mapping[f.name], type=f.type, nullable=f.nullable)
|
|
207
|
+
for f in spec.fields
|
|
208
|
+
),
|
|
209
|
+
key=tuple(mapping[k] for k in spec.key),
|
|
210
|
+
pii=tuple(mapping[p] for p in spec.pii),
|
|
211
|
+
# Residency is not an identifier. It is a jurisdiction, it is a hard constraint on
|
|
212
|
+
# placement, and hashing it would make the constraint unenforceable.
|
|
213
|
+
residency=spec.residency,
|
|
214
|
+
)
|
|
215
|
+
)
|
|
216
|
+
|
|
217
|
+
relations: list[RelationSpec] = []
|
|
218
|
+
for relation in model.relations:
|
|
219
|
+
hashed = _digest(salt, _RELATION_PREFIX, relation.source, relation.name)
|
|
220
|
+
relation_names.setdefault(relation.source, {})[relation.name] = hashed
|
|
221
|
+
relations.append(
|
|
222
|
+
RelationSpec(
|
|
223
|
+
name=hashed,
|
|
224
|
+
source=entity_names[relation.source],
|
|
225
|
+
target=entity_names[relation.target],
|
|
226
|
+
)
|
|
227
|
+
)
|
|
228
|
+
|
|
229
|
+
atomic = tuple(
|
|
230
|
+
tuple(sorted(entity_names[member] for member in group)) for group in model.atomic
|
|
231
|
+
)
|
|
232
|
+
|
|
233
|
+
hashed_model = assemble(
|
|
234
|
+
entities=tuple(entities),
|
|
235
|
+
relations=tuple(relations),
|
|
236
|
+
atomic=tuple(sorted(atomic)),
|
|
237
|
+
# The cost ceiling is a number and a currency, not an identifier.
|
|
238
|
+
cost_ceiling=model.cost_ceiling,
|
|
239
|
+
)
|
|
240
|
+
return hashed_model, NameMap(
|
|
241
|
+
entities=entity_names, fields=field_names, relations=relation_names
|
|
242
|
+
)
|