smart-data-engine-sdk 0.1.0__py3-none-any.whl

This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
Files changed (73) hide show
  1. sde/__init__.py +318 -0
  2. sde/_cutover_project.py +179 -0
  3. sde/_local_state.py +188 -0
  4. sde/_operator_deadline.py +50 -0
  5. sde/_usage.py +314 -0
  6. sde/bulk.py +79 -0
  7. sde/canonical.py +141 -0
  8. sde/capabilities.py +62 -0
  9. sde/cutover.py +286 -0
  10. sde/engines/__init__.py +0 -0
  11. sde/engines/_clickhouse_connection.py +224 -0
  12. sde/engines/_index_build.py +294 -0
  13. sde/engines/_operator.py +394 -0
  14. sde/engines/_staging.py +222 -0
  15. sde/engines/_storage.py +22 -0
  16. sde/engines/_write_fences.py +271 -0
  17. sde/engines/clickhouse.py +1115 -0
  18. sde/engines/orderbook.py +457 -0
  19. sde/engines/postgres.py +967 -0
  20. sde/entity.py +170 -0
  21. sde/errors.py +103 -0
  22. sde/explain.py +300 -0
  23. sde/frozen_verification.py +152 -0
  24. sde/generation.py +131 -0
  25. sde/groups.py +97 -0
  26. sde/hashing.py +242 -0
  27. sde/index_build.py +313 -0
  28. sde/index_operator.py +347 -0
  29. sde/infer.py +461 -0
  30. sde/inspection.py +62 -0
  31. sde/internal.py +90 -0
  32. sde/layout.py +669 -0
  33. sde/local_cutover.py +801 -0
  34. sde/logging.py +143 -0
  35. sde/migration.py +856 -0
  36. sde/model.py +482 -0
  37. sde/physical.py +531 -0
  38. sde/placement.py +1010 -0
  39. sde/provisioning.py +63 -0
  40. sde/py.typed +0 -0
  41. sde/query.py +521 -0
  42. sde/routing.py +85 -0
  43. sde/schema.py +466 -0
  44. sde/session.py +993 -0
  45. sde/shapes.py +153 -0
  46. sde/staging.py +264 -0
  47. sde/staging_operator.py +393 -0
  48. sde/telemetry.py +1087 -0
  49. sde/testing/__init__.py +14 -0
  50. sde/testing/loader.py +175 -0
  51. sde/testing/memory.py +331 -0
  52. sde/types.py +228 -0
  53. sde/verification.py +220 -0
  54. sde/watermark.py +222 -0
  55. sde/write_fence.py +283 -0
  56. sde_demo/__init__.py +1 -0
  57. sde_demo/__main__.py +183 -0
  58. sde_demo/diagnostics.py +92 -0
  59. sde_demo/model.py +75 -0
  60. sde_demo/project.py +312 -0
  61. sde_demo/py.typed +0 -0
  62. sde_demo/query_count.py +301 -0
  63. sde_demo/resources.py +969 -0
  64. sde_demo/runtime.py +419 -0
  65. sde_demo/verification.py +242 -0
  66. sde_operator/__init__.py +1 -0
  67. sde_operator/__main__.py +210 -0
  68. smart_data_engine_sdk-0.1.0.dist-info/METADATA +174 -0
  69. smart_data_engine_sdk-0.1.0.dist-info/RECORD +73 -0
  70. smart_data_engine_sdk-0.1.0.dist-info/WHEEL +4 -0
  71. smart_data_engine_sdk-0.1.0.dist-info/entry_points.txt +3 -0
  72. smart_data_engine_sdk-0.1.0.dist-info/licenses/LICENSE +201 -0
  73. smart_data_engine_sdk-0.1.0.dist-info/licenses/NOTICE +13 -0
sde_demo/runtime.py ADDED
@@ -0,0 +1,419 @@
1
+ """A bounded logical workload with explicit uncertain writes and value-free reports."""
2
+
3
+ from __future__ import annotations
4
+
5
+ import re
6
+ import time
7
+ from collections.abc import Callable, Mapping
8
+ from decimal import Decimal
9
+ from functools import partial
10
+ from pathlib import Path
11
+ from typing import Any, TypeVar
12
+ from uuid import uuid4
13
+
14
+ import sde
15
+
16
+ from .model import (
17
+ BASE_TIME,
18
+ CELSIUS_BASE_CENTS,
19
+ CELSIUS_MODULUS,
20
+ GENERATOR_ID,
21
+ HUMIDITY_BASE,
22
+ HUMIDITY_MODULUS,
23
+ WORKLOADS,
24
+ model,
25
+ reading,
26
+ )
27
+ from .project import (
28
+ DemoRefused,
29
+ config,
30
+ credentials,
31
+ engine,
32
+ public_keys,
33
+ read,
34
+ require_drivers,
35
+ write,
36
+ )
37
+
38
+ T = TypeVar("T")
39
+ FLEET_PAGE = 100
40
+ MAX_RUN_ROWS = 10000
41
+ """The fleet checks one cross-station page exactly; its count and sum cover the whole window."""
42
+ ALERT_HUMIDITY = 95
43
+ """The alert threshold. Humidity is 30 + sequence % 70, so a reading alerts when sequence % 70 is
44
+ 65 or more - five readings in seventy, known exactly from the sequence number."""
45
+ ALERT_PAGE = 100
46
+
47
+
48
+ def _failed_before_writing(report: Mapping[str, Any]) -> bool:
49
+ """A run that stopped, failed, before it recorded its first write: it wrote no row."""
50
+ return (
51
+ report.get("status") == "incomplete"
52
+ and report.get("pending", False) is None
53
+ and report.get("generator_id") == GENERATOR_ID
54
+ and all(
55
+ type(report.get(field)) is int and report[field] == 0
56
+ for field in ("acknowledged_rows", "verified_after_uncertain_rows", "verified_rows")
57
+ )
58
+ )
59
+
60
+
61
+ def fleet_runs(root: Path, *, project_id: str) -> dict[str, int]:
62
+ """Every earlier run of this project, by run ID, with the rows it verified.
63
+
64
+ Fleet analytics reads every station's rows in one time window, and every run in a directory
65
+ writes the same timeline, so the exact answer is a sum over the runs. It is known exactly:
66
+ generated values depend only on the sequence number, and each run's local report says how many
67
+ rows it verified. A run that did not complete - interrupted, or with an uncertain batch - makes
68
+ that sum unknowable, so the workload refuses instead of comparing a read with a guess.
69
+
70
+ One unfinished run is known exactly: one that failed before its first write. A run records the
71
+ batch it is about to write before writing it, so a failed run with no pending batch and no
72
+ acknowledged or verified row wrote nothing, and it adds nothing to the sum. Without this, a
73
+ first run that could not open its session - a missing driver - stopped every later fleet run
74
+ in that directory.
75
+ """
76
+ runs: dict[str, int] = {}
77
+ for path in sorted((root / "runs").glob("*/report.json")):
78
+ report = read(path)
79
+ if report.get("project_id") != project_id:
80
+ continue
81
+ identity = report.get("run_id")
82
+ if _failed_before_writing(report) and isinstance(identity, str) and (
83
+ path.parent.name == identity
84
+ ):
85
+ continue
86
+ if (
87
+ not isinstance(identity, str)
88
+ or re.fullmatch(r"[0-9a-f]{32}", identity) is None
89
+ or path.parent.name != identity
90
+ or report.get("status") != "complete"
91
+ or report.get("pending", False) is not None
92
+ or report.get("generator_id") != GENERATOR_ID
93
+ or type(report.get("verified_rows")) is not int
94
+ or not 0 <= report["verified_rows"] <= MAX_RUN_ROWS
95
+ ):
96
+ raise DemoRefused(
97
+ "Fleet analytics reads every run's rows, so every earlier run of this project must "
98
+ "be complete; inspect the unfinished run first."
99
+ )
100
+ runs[identity] = report["verified_rows"]
101
+ return runs
102
+
103
+
104
+ def fleet_expected(
105
+ runs: Mapping[str, int], through: int, limit: int
106
+ ) -> tuple[list[dict[str, Any]], int, Decimal]:
107
+ """The first page, row count and celsius total of the window holding sequences 1..through.
108
+
109
+ A page is in key order, station then time, and a station is its run's namespace, so the page
110
+ walks runs in the order of their stations and each run's rows in sequence order.
111
+ """
112
+ page: list[dict[str, Any]] = []
113
+ total, cents = 0, 0
114
+ for identity in sorted(runs, key=lambda item: reading(item, 0, 1)["station"]):
115
+ rows = min(through, runs[identity])
116
+ total += rows
117
+ for sequence in range(1, rows + 1):
118
+ cents += CELSIUS_BASE_CENTS + sequence % CELSIUS_MODULUS
119
+ if len(page) < limit:
120
+ page.append(reading(identity, 0, sequence))
121
+ return page, total, Decimal(f"{cents // 100}.{cents % 100:02d}")
122
+
123
+
124
+ def alert_expected(
125
+ run_id: str, through: int, limit: int
126
+ ) -> tuple[list[dict[str, Any]], int, Decimal | None]:
127
+ """The first page, count and celsius total of this run's alerts among sequences 1..through.
128
+
129
+ One station, so key order - station, then time - is sequence order. With no alert yet the total
130
+ is None, as the library reports a summary of no values on every engine.
131
+ """
132
+ page: list[dict[str, Any]] = []
133
+ total, cents = 0, 0
134
+ for sequence in range(1, through + 1):
135
+ if HUMIDITY_BASE + sequence % HUMIDITY_MODULUS < ALERT_HUMIDITY:
136
+ continue
137
+ total += 1
138
+ cents += CELSIUS_BASE_CENTS + sequence % CELSIUS_MODULUS
139
+ if len(page) < limit:
140
+ page.append(reading(run_id, 0, sequence))
141
+ return page, total, Decimal(f"{cents // 100}.{cents % 100:02d}") if total else None
142
+
143
+
144
+ def limits(iterations: int, batch_size: int, interval_ms: int, recovery_ms: int) -> None:
145
+ if (
146
+ any(type(value) is not int for value in (iterations, batch_size, interval_ms, recovery_ms))
147
+ or not 1 <= iterations <= 1000
148
+ or not 1 <= batch_size <= 1000
149
+ or iterations * batch_size > MAX_RUN_ROWS
150
+ or not 0 <= interval_ms <= 1000
151
+ or not 0 <= recovery_ms <= 30000
152
+ ):
153
+ raise DemoRefused(
154
+ "Use 1-1000 iterations/batch, at most 10000 rows, interval 0-1000 ms "
155
+ "and recovery 0-30000 ms."
156
+ )
157
+
158
+
159
+ def run(
160
+ root: Path,
161
+ *,
162
+ iterations: int = 10,
163
+ batch_size: int = 10,
164
+ interval_ms: int = 100,
165
+ recovery_ms: int = 10000,
166
+ workload: str = "mixed",
167
+ ) -> dict[str, Any]:
168
+ limits(iterations, batch_size, interval_ms, recovery_ms)
169
+ if workload not in WORKLOADS:
170
+ raise DemoRefused("Choose the mixed, point, analytics, fleet or alerts workload.")
171
+ settings = config(root)
172
+ require_drivers(binding["dialect"] for binding in settings["engines"].values())
173
+ # Read before this run's own report exists; a run that starts later is not in the window.
174
+ earlier = fleet_runs(root, project_id=settings["project_id"]) if workload == "fleet" else {}
175
+ logical, keys = model(), public_keys(settings["public_keys"])
176
+ dsns = credentials(root, "runtime", settings["engines"])
177
+ factories = {
178
+ name: (lambda dialect=binding["dialect"], dsn=dsns[name]: engine(dialect, dsn))
179
+ for name, binding in settings["engines"].items()
180
+ }
181
+ run_id = uuid4().hex
182
+ directory = root / "runs" / run_id
183
+ recorder = sde.Recorder(logical.version)
184
+ session: sde.Session | None = None
185
+ fingerprint: str | None = None
186
+ report: dict[str, Any] = {
187
+ "protocol": 2,
188
+ "generator_id": GENERATOR_ID,
189
+ "run_id": run_id,
190
+ "language": "python",
191
+ "status": "running",
192
+ "workload": workload,
193
+ "sdk_version": sde.__version__,
194
+ "sdk_module": sde.__file__,
195
+ "project_id": settings["project_id"],
196
+ "model_version": logical.version,
197
+ "acknowledged_rows": 0,
198
+ "verified_after_uncertain_rows": 0,
199
+ "verified_rows": 0,
200
+ "pending": None,
201
+ "map_versions": [],
202
+ "read_retries": 0,
203
+ }
204
+
205
+ def checkpoint() -> None:
206
+ write(directory / "report.json", report)
207
+
208
+ def opened(*, fresh: bool = False) -> sde.Session:
209
+ nonlocal session, fingerprint
210
+ if (root / "reset-request.json").exists():
211
+ raise DemoRefused("Reset was requested; this workload has stopped.")
212
+ placement = sde.load_local_map(
213
+ root / "state", model=logical, project_id=settings["project_id"], public_key=keys
214
+ )
215
+ if fresh or session is None or placement.fingerprint != fingerprint:
216
+ if session is not None:
217
+ session.close()
218
+ session = None
219
+ required = {
220
+ material.engine for group in placement.groups.values() for material in group.all()
221
+ }
222
+ if required - factories.keys():
223
+ raise DemoRefused("An active materialization has no local binding.")
224
+ active_factories = {name: factories[name] for name in sorted(required)}
225
+ session = sde.Session.connect(
226
+ logical,
227
+ placement,
228
+ active_factories,
229
+ recorder=recorder,
230
+ project_id=settings["project_id"],
231
+ )
232
+ fingerprint = placement.fingerprint
233
+ if placement.map_version not in report["map_versions"]:
234
+ report["map_versions"].append(placement.map_version)
235
+ return session
236
+
237
+ def read_retry(action: Callable[[sde.Session], T]) -> T:
238
+ deadline = time.monotonic() + recovery_ms / 1000
239
+ fresh = False
240
+ while True:
241
+ try:
242
+ return action(opened(fresh=fresh))
243
+ except (sde.EngineError, sde.MapRolledBack, sde.MigrationRefused):
244
+ if time.monotonic() >= deadline:
245
+ raise
246
+ report["read_retries"] += 1
247
+ fresh = True
248
+ time.sleep(0.05)
249
+
250
+ def same(actual: Mapping[str, Any] | None, expected: dict[str, Any]) -> None:
251
+ if actual != expected:
252
+ raise DemoRefused("A logical read did not match this run's synthetic input.")
253
+
254
+ def check_rows(client: sde.Session, rows: list[dict[str, Any]]) -> bool:
255
+ for row in rows:
256
+ actual = client.get(
257
+ "WeatherReading", {key: row[key] for key in ("station", "at")}, fresh=True
258
+ )
259
+ if actual is None:
260
+ return False
261
+ same(actual, row)
262
+ return True
263
+
264
+ def point(current: sde.Session, row: dict[str, Any]) -> Any:
265
+ return current.get("WeatherReading", {key: row[key] for key in ("station", "at")})
266
+
267
+ def counted(
268
+ current: sde.Session, where: dict[str, Any], bounds: sde.Range | None = None
269
+ ) -> int:
270
+ return current.count("WeatherReading", where=where, bounds=bounds)
271
+
272
+ def summarized(
273
+ current: sde.Session, where: dict[str, Any], bounds: sde.Range | None = None
274
+ ) -> Any:
275
+ return current.summarize("WeatherReading", "celsius", where=where, bounds=bounds)
276
+
277
+ checkpoint()
278
+ began = time.monotonic_ns()
279
+ try:
280
+ # The group's size at the start of the run and again at its end, for the window: the
281
+ # engine's catalogue answers with numbers, and a refused read leaves the size unknown.
282
+ read_retry(lambda current: current.measure_storage())
283
+ for iteration in range(iterations):
284
+ first = iteration * batch_size + 1
285
+ rows = [reading(run_id, 0, number) for number in range(first, first + batch_size)]
286
+ # Opening/refresh errors happen before the write intent and may be retried safely.
287
+ client = read_retry(lambda current: current)
288
+ report["pending"] = {"first": first, "count": batch_size}
289
+ checkpoint()
290
+ try:
291
+ client.save_many("WeatherReading", rows)
292
+ except sde.EngineError:
293
+ # No replay. A visible exact batch establishes the result; an absent or partial
294
+ # batch does not prove rollback. Leave its durable range pending on refusal.
295
+ deadline = time.monotonic() + recovery_ms / 1000
296
+ while True:
297
+ try:
298
+ if check_rows(opened(fresh=True), rows):
299
+ report["verified_after_uncertain_rows"] += batch_size
300
+ break
301
+ except (sde.EngineError, sde.MapRolledBack, sde.MigrationRefused):
302
+ pass
303
+ if time.monotonic() >= deadline:
304
+ raise DemoRefused(
305
+ "Write outcome is uncertain. Inspect this run's pending "
306
+ "range locally; the starter did not replay it."
307
+ ) from None
308
+ time.sleep(0.05)
309
+ else:
310
+ report["acknowledged_rows"] += batch_size
311
+ report["pending"] = None
312
+ checkpoint()
313
+ last = rows[-1]
314
+ repeats = 20 if workload == "point" else 1
315
+ for _ in range(repeats):
316
+ same(read_retry(partial(point, row=last)), last)
317
+ count = first + batch_size - 1
318
+ where = {"station": last["station"]}
319
+ if workload == "fleet":
320
+ # Every station, one time window: the traffic a time-first layout is for. The
321
+ # expected answer is exact, from this directory's run reports (``fleet_runs``).
322
+ fleet_page, fleet_rows, fleet_celsius = fleet_expected(
323
+ {**earlier, run_id: count}, count, FLEET_PAGE
324
+ )
325
+ bounds = sde.Range(
326
+ "at", low=reading(run_id, 0, 1)["at"], high=reading(run_id, 0, count + 1)["at"]
327
+ )
328
+
329
+ def check_fleet_page(
330
+ current: sde.Session,
331
+ bounds: sde.Range = bounds,
332
+ expected: list[dict[str, Any]] = fleet_page,
333
+ ) -> None:
334
+ page = current.scan("WeatherReading", bounds=bounds, limit=len(expected))
335
+ if list(page.rows) != expected:
336
+ raise DemoRefused("The fleet page did not match this directory's runs.")
337
+
338
+ read_retry(check_fleet_page)
339
+ if read_retry(partial(counted, where={}, bounds=bounds)) != fleet_rows:
340
+ raise DemoRefused("The fleet count did not match this directory's runs.")
341
+ summary = read_retry(partial(summarized, where={}, bounds=bounds))
342
+ if summary.count != fleet_rows or summary.total != fleet_celsius:
343
+ raise DemoRefused(
344
+ "The exact fleet summary did not match this directory's runs."
345
+ )
346
+ elif workload == "alerts":
347
+ # One station's readings at or above the alert threshold: a range on humidity, a
348
+ # field outside the key, which the key cannot serve and an index can. The expected
349
+ # answer is exact, from the generator (``alert_expected``).
350
+ alert_page, alert_rows, alert_celsius = alert_expected(run_id, count, ALERT_PAGE)
351
+ alert_bounds = sde.Range("humidity", low=ALERT_HUMIDITY)
352
+
353
+ def check_alert_page(
354
+ current: sde.Session,
355
+ where: dict[str, Any] = where,
356
+ bounds: sde.Range = alert_bounds,
357
+ expected: list[dict[str, Any]] = alert_page,
358
+ ) -> None:
359
+ page = current.scan(
360
+ "WeatherReading", where=where, bounds=bounds, limit=ALERT_PAGE
361
+ )
362
+ if list(page.rows) != expected:
363
+ raise DemoRefused("The alert page did not match this run.")
364
+
365
+ read_retry(check_alert_page)
366
+ if read_retry(partial(counted, where=where, bounds=alert_bounds)) != alert_rows:
367
+ raise DemoRefused("The alert count did not match this run.")
368
+ summary = read_retry(partial(summarized, where=where, bounds=alert_bounds))
369
+ if summary.count != alert_rows or summary.total != alert_celsius:
370
+ raise DemoRefused("The exact alert summary did not match this run.")
371
+ elif workload != "point" or iteration == iterations - 1:
372
+
373
+ def check_page(
374
+ current: sde.Session, where: dict[str, Any] = where, count: int = count
375
+ ) -> None:
376
+ page = current.scan(
377
+ "WeatherReading",
378
+ where=where,
379
+ bounds=sde.Range("at", low=BASE_TIME),
380
+ limit=min(count, 1000),
381
+ )
382
+ expected = [reading(run_id, 0, index) for index in range(1, len(page.rows) + 1)]
383
+ if list(page.rows) != expected or len(page.rows) != min(count, 1000):
384
+ raise DemoRefused("The bounded logical page did not match this run.")
385
+
386
+ read_retry(check_page)
387
+ if read_retry(partial(counted, where=where)) != count:
388
+ raise DemoRefused("The logical count did not match this run.")
389
+ summary = read_retry(partial(summarized, where=where))
390
+ cents = sum(1525 + index % 1000 for index in range(1, count + 1))
391
+ total = Decimal(f"{cents // 100}.{cents % 100:02d}")
392
+ if summary.count != count or summary.total != total:
393
+ raise DemoRefused("The exact logical summary did not match this run.")
394
+ report["verified_rows"] = count
395
+ checkpoint()
396
+ if interval_ms and iteration + 1 < iterations:
397
+ time.sleep(interval_ms / 1000)
398
+ read_retry(lambda current: current.measure_storage())
399
+ report["status"] = "complete"
400
+ except BaseException as exc:
401
+ report["status"] = "incomplete"
402
+ report["failure"] = type(exc).__name__
403
+ raise
404
+ finally:
405
+ report["elapsed_ns"] = str(time.monotonic_ns() - began)
406
+ try:
407
+ if session is not None:
408
+ session.close()
409
+ except BaseException:
410
+ report["status"] = "incomplete"
411
+ report["cleanup_failed"] = True
412
+ raise
413
+ finally:
414
+ window = recorder.roll()
415
+ if window is not None:
416
+ write(directory / "window.json", window.as_record(logical))
417
+ recorder.acknowledge(1)
418
+ checkpoint()
419
+ return report
@@ -0,0 +1,242 @@
1
+ """Verify completed local Weather runs against the exact active source, without exposing rows."""
2
+
3
+ from __future__ import annotations
4
+
5
+ import hashlib
6
+ import re
7
+ from collections.abc import Iterator, Mapping, Sequence
8
+ from contextlib import contextmanager
9
+ from dataclasses import dataclass
10
+ from pathlib import Path
11
+ from typing import Any
12
+
13
+ import sde
14
+ from sde.generation import GENERATIONS_SINCE, check_map_project
15
+
16
+ from .model import GENERATOR_ID, model, reading
17
+ from .project import DemoRefused, config, credentials, decode, engine, payload, public_keys
18
+
19
+ MAX_RUNS = 32
20
+ MAX_RUN_ROWS = 10_000
21
+ MAX_TOTAL_ROWS = 100_000
22
+ PAGE_ROWS = 1000
23
+
24
+
25
+ @dataclass(frozen=True)
26
+ class CompletedRun:
27
+ """A validated local report descriptor; expected values are regenerated only in this process."""
28
+
29
+ run_id: str
30
+ language: str
31
+ expected_rows: int
32
+ report_sha256: str
33
+
34
+
35
+ @contextmanager
36
+ def _refusals() -> Iterator[None]:
37
+ try:
38
+ yield
39
+ except DemoRefused:
40
+ raise
41
+ except Exception:
42
+ # Driver/decoder messages can contain credentials or values returned by an engine.
43
+ raise DemoRefused(
44
+ "Local run verification did not complete; preserve the reports and inspect locally."
45
+ ) from None
46
+
47
+
48
+ def _selected(run_ids: Sequence[str]) -> tuple[str, ...]:
49
+ if (
50
+ isinstance(run_ids, (str, bytes, bytearray))
51
+ or not isinstance(run_ids, Sequence)
52
+ or not 1 <= len(run_ids) <= MAX_RUNS
53
+ ):
54
+ raise DemoRefused("Select between 1 and 32 completed run IDs.")
55
+ if any(
56
+ not isinstance(identity, str) or re.fullmatch(r"[0-9a-f]{32}", identity) is None
57
+ for identity in run_ids
58
+ ):
59
+ raise DemoRefused("Each run ID must be its original lowercase hexadecimal identifier.")
60
+ if len(set(run_ids)) != len(run_ids):
61
+ raise DemoRefused("Select each run ID only once.")
62
+ return tuple(run_ids)
63
+
64
+
65
+ def _counter(value: object) -> int:
66
+ if type(value) is not int or value < 0:
67
+ raise DemoRefused("Completed run counters must be nonnegative integers, without coercion.")
68
+ return value
69
+
70
+
71
+ def _load_completed(
72
+ root: Path, run_ids: Sequence[str], *, project_id: str, model_version: str
73
+ ) -> tuple[CompletedRun, ...]:
74
+ result = []
75
+ total = 0
76
+ for identity in _selected(run_ids):
77
+ data = payload(root / "runs" / identity / "report.json")
78
+ report = decode(data)
79
+ if (
80
+ type(report.get("protocol")) is not int
81
+ or report["protocol"] != 2
82
+ or report.get("generator_id") != GENERATOR_ID
83
+ ):
84
+ raise DemoRefused("Verification requires report protocol 2 and its known generator ID.")
85
+ if (
86
+ report.get("run_id") != identity
87
+ or report.get("project_id") != project_id
88
+ or report.get("model_version") != model_version
89
+ or report.get("language") not in ("python", "typescript")
90
+ ):
91
+ raise DemoRefused("A run report belongs to another run, project, model or language.")
92
+ if (
93
+ report.get("status") != "complete"
94
+ or "pending" not in report
95
+ or report["pending"] is not None
96
+ or report.get("cleanup_failed", False) is not False
97
+ or report.get("failure") is not None
98
+ ):
99
+ raise DemoRefused(
100
+ "Only completed runs without pending work or cleanup failure can be verified."
101
+ )
102
+ acknowledged, uncertain, verified = (
103
+ _counter(report.get(name))
104
+ for name in ("acknowledged_rows", "verified_after_uncertain_rows", "verified_rows")
105
+ )
106
+ if acknowledged + uncertain != verified:
107
+ raise DemoRefused(
108
+ "A completed run's acknowledged and independently verified totals disagree."
109
+ )
110
+ if verified > MAX_RUN_ROWS:
111
+ raise DemoRefused("Verification permits at most 10000 expected rows per run.")
112
+ total += verified
113
+ if total > MAX_TOTAL_ROWS:
114
+ raise DemoRefused("Verification permits at most 100000 expected rows per invocation.")
115
+ result.append(
116
+ CompletedRun(identity, report["language"], verified, hashlib.sha256(data).hexdigest())
117
+ )
118
+ return tuple(result)
119
+
120
+
121
+ def load_completed_runs(root: Path, run_ids: Sequence[str]) -> tuple[CompletedRun, ...]:
122
+ """Load bounded completed reports locally; query checks can sum each expected_rows value."""
123
+ with _refusals():
124
+ settings = config(root)
125
+ return _load_completed(
126
+ root, run_ids, project_id=settings["project_id"], model_version=model().version
127
+ )
128
+
129
+
130
+ def _active_map(
131
+ root: Path, settings: Mapping[str, Any], logical: sde.LogicalModel
132
+ ) -> sde.PlacementMap:
133
+ if (root / "reset-request.json").exists():
134
+ raise DemoRefused("Reset was requested; run verification has stopped.")
135
+ document = decode(payload(root / "state" / "active-map.json"))
136
+ placement = sde.load_map(
137
+ document,
138
+ model=logical,
139
+ public_key=public_keys(settings["public_keys"]),
140
+ require_signature=True,
141
+ )
142
+ if placement.contract < GENERATIONS_SINCE:
143
+ raise DemoRefused("Run verification requires an active generation-bearing Weather map.")
144
+ check_map_project(placement, settings["project_id"])
145
+ return placement
146
+
147
+
148
+ def _map_identity(placement: sde.PlacementMap) -> tuple[str | None, str, int, str | None]:
149
+ # A trusted re-signature or JSON formatting change does not change the instructions.
150
+ return (
151
+ placement.project_id,
152
+ placement.model_version,
153
+ placement.map_version,
154
+ placement.fingerprint,
155
+ )
156
+
157
+
158
+ def _verify_one(client: sde.Session, completed: CompletedRun) -> None:
159
+ station = reading(completed.run_id, 0, 1)["station"]
160
+ seen = 0
161
+ after: Mapping[str, Any] | None = None
162
+ while True:
163
+ page = client.scan(
164
+ "WeatherReading",
165
+ where={"station": station},
166
+ after=after,
167
+ limit=min(PAGE_ROWS, completed.expected_rows - seen + 1),
168
+ fresh=True,
169
+ )
170
+ for actual in page.rows:
171
+ if seen >= completed.expected_rows:
172
+ raise DemoRefused("A completed run's namespace contains additional source rows.")
173
+ expected = reading(completed.run_id, 0, seen + 1)
174
+ if actual != expected:
175
+ raise DemoRefused("A completed run's exact values do not match the active source.")
176
+ seen += 1
177
+ if page.next_after is None:
178
+ break
179
+ if not page.rows or dict(page.next_after) != {
180
+ key: page.rows[-1][key] for key in ("station", "at")
181
+ }:
182
+ raise DemoRefused("The source returned an inconsistent page cursor.")
183
+ after = page.next_after
184
+ if seen != completed.expected_rows:
185
+ raise DemoRefused("A completed run is missing expected source rows.")
186
+
187
+
188
+ def verify_runs(root: Path, run_ids: Sequence[str]) -> dict[str, Any]:
189
+ """Read all expected values and reject extras in each run's namespace, using runtime rights."""
190
+ with _refusals():
191
+ settings, logical = config(root), model()
192
+ completed = _load_completed(
193
+ root, run_ids, project_id=settings["project_id"], model_version=logical.version
194
+ )
195
+ placement = _active_map(root, settings, logical)
196
+ identity = _map_identity(placement)
197
+ needed = {
198
+ material.engine for group in placement.groups.values() for material in group.all()
199
+ }
200
+ if needed - settings["engines"].keys():
201
+ raise DemoRefused("An active materialization has no local runtime binding.")
202
+ dsns = credentials(root, "runtime", settings["engines"])
203
+ factories = {
204
+ name: (
205
+ lambda dialect=settings["engines"][name]["dialect"], dsn=dsns[name]: engine(
206
+ dialect, dsn
207
+ )
208
+ )
209
+ for name in sorted(needed)
210
+ }
211
+ with sde.Session.connect(
212
+ logical, placement, factories, project_id=settings["project_id"]
213
+ ) as client:
214
+ for completed_run in completed:
215
+ _verify_one(client, completed_run)
216
+ if _map_identity(_active_map(root, settings, logical)) != identity:
217
+ raise DemoRefused(
218
+ "The active map changed during run verification; repeat against the current source."
219
+ )
220
+ for completed_run in completed:
221
+ data = payload(root / "runs" / completed_run.run_id / "report.json")
222
+ if hashlib.sha256(data).hexdigest() != completed_run.report_sha256:
223
+ raise DemoRefused("A completed report changed during run verification.")
224
+ return {
225
+ "protocol": 1,
226
+ "status": "verified",
227
+ "project_id": settings["project_id"],
228
+ "model_version": logical.version,
229
+ "generator_id": GENERATOR_ID,
230
+ "map_version": placement.map_version,
231
+ "map_fingerprint": placement.fingerprint,
232
+ "verified_rows": sum(item.expected_rows for item in completed),
233
+ "runs": [
234
+ {
235
+ "run_id": item.run_id,
236
+ "language": item.language,
237
+ "verified_rows": item.expected_rows,
238
+ "report_sha256": item.report_sha256,
239
+ }
240
+ for item in completed
241
+ ],
242
+ }
@@ -0,0 +1 @@
1
+ """Dedicated local operator command, separate from the silent application SDK."""