smart-data-engine-sdk 0.1.0.dev0__py3-none-any.whl

This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
sde/telemetry.py ADDED
@@ -0,0 +1,736 @@
1
+ """Measuring what the application actually does, without ever seeing what it does it to.
2
+
3
+ This is the input the placement decision is made from, so its shape matters more than its precision.
4
+ Three constraints shaped everything here.
5
+
6
+ **It carries no values.** A record is keyed by an operation shape, which is assembled from the
7
+ structure of a call and never sees its arguments. There is no code path by which a customer's row
8
+ reaches a telemetry record, which is why this file can be read by a client and believed.
9
+
10
+ **It cannot cost anything.** Routing already has a one percent budget for the whole library, and
11
+ recording happens on the same path. So: no locks on the hot path per record, no string formatting,
12
+ no stack walking except once per shape, and a histogram rather than a list of samples.
13
+
14
+ **It cannot fail the caller.** Every entry point is wrapped in :func:`~sde.internal.guard`. A bug in
15
+ an aggregation counter must not take down somebody's request - and the failure is counted, so it is
16
+ not invisible either.
17
+
18
+ The histogram deserves a word, because it is the one deliberate loss of precision. Latency lands in
19
+ exponential buckets, so a percentile read out of it is approximate - within one bucket width, which
20
+ is a factor of two at the extremes. That is ample for the decision it feeds: the planner cares
21
+ whether a group's reads are microseconds or milliseconds, not whether p99 is 412 or 431
22
+ microseconds. Keeping exact samples would mean either unbounded memory or reservoir sampling, and
23
+ reservoir sampling gets the tail wrong in exactly the region the planner looks at.
24
+ """
25
+
26
+ from __future__ import annotations
27
+
28
+ import math
29
+ import sys
30
+ import threading
31
+ from collections import deque
32
+ from collections.abc import Collection, Mapping, Sequence
33
+ from dataclasses import dataclass, field, replace
34
+ from dataclasses import fields as dataclass_fields
35
+ from typing import Any
36
+
37
+ from .groups import Group, colocation_groups
38
+ from .internal import guard
39
+ from .logging import log
40
+ from .model import LogicalModel
41
+ from .shapes import WRITE_KINDS
42
+
43
+ __all__ = [
44
+ "MEASURED_FIELDS",
45
+ "CopyFreshness",
46
+ "FanOutStats",
47
+ "GroupFeatures",
48
+ "Histogram",
49
+ "Recorder",
50
+ "ShapeStats",
51
+ "Window",
52
+ "has_time_dimension",
53
+ ]
54
+
55
+ # 1 µs to about 17 s, doubling. Twenty-five buckets is enough to tell a cache hit from a full scan,
56
+ # which is the distinction the planner actually acts on.
57
+ BUCKET_COUNT = 25
58
+ BUCKET_BASE_NS = 1_000
59
+
60
+
61
+ class Histogram:
62
+ """Exponential-bucket histogram. Fixed memory, O(1) record, approximate percentiles."""
63
+
64
+ __slots__ = ("buckets", "count", "total")
65
+
66
+ def __init__(self) -> None:
67
+ self.buckets = [0] * BUCKET_COUNT
68
+ self.count = 0
69
+ self.total = 0
70
+
71
+ def record(self, nanoseconds: int) -> None:
72
+ """Put one duration in its bucket. Integer arithmetic only, and that is the point.
73
+
74
+ This used to read ``int(math.log2(nanoseconds / BUCKET_BASE_NS)) + 1``, which is the same
75
+ function and the wrong way to compute it once a second language has to agree. ``log2`` is
76
+ not required by IEEE 754 to be correctly rounded, so two libm implementations may differ in
77
+ the last bit - and one bit at a power-of-two boundary is a different bucket, which is a
78
+ different p99 for identical traffic. The bit length of the integer quotient is exact
79
+ everywhere, and it is cheaper on a path that runs per operation.
80
+
81
+ Verified rather than asserted: the two forms were compared over every value below 40 000,
82
+ every power-of-two boundary and its neighbours, and 400 000 random durations up to 10^13 ns.
83
+ Zero disagreements.
84
+
85
+ **And the vectors cannot see the difference, which is the point of saying so here.** An
86
+ earlier version of this note claimed ``telemetry/002`` pins it. Measured: replacing this
87
+ with the logarithm form passes every vector, because glibc's ``log2`` and V8's are both
88
+ exact at a power of two - so the two runtimes we have agree, and the hazard is a *third*
89
+ libm that does not. A property no output can distinguish on the machines available is not
90
+ one a vector can hold, so it is held statically instead: see
91
+ ``test_the_bucket_index_never_reaches_for_a_logarithm``.
92
+ """
93
+ self.count += 1
94
+ self.total += nanoseconds
95
+ if nanoseconds < BUCKET_BASE_NS:
96
+ self.buckets[0] += 1
97
+ return
98
+ index: int = min(BUCKET_COUNT - 1, (nanoseconds // BUCKET_BASE_NS).bit_length())
99
+ self.buckets[index] += 1
100
+
101
+ def percentile_ms(self, fraction: float) -> float | None:
102
+ """Approximate percentile in milliseconds, or None if nothing was recorded.
103
+
104
+ Returns the *upper* edge of the bucket the percentile falls in. Rounding up rather than
105
+ interpolating is deliberate: a placement decision made on an optimistic latency figure is
106
+ the wrong kind of wrong.
107
+ """
108
+ if self.count == 0:
109
+ return None
110
+ target = fraction * self.count
111
+ seen = 0
112
+ for index, hits in enumerate(self.buckets):
113
+ seen += hits
114
+ if seen >= target:
115
+ upper_ns: int = BUCKET_BASE_NS * (2**index)
116
+ return upper_ns / 1_000_000
117
+ return None
118
+
119
+ def merge(self, other: Histogram) -> None:
120
+ for index, hits in enumerate(other.buckets):
121
+ self.buckets[index] += hits
122
+ self.count += other.count
123
+ self.total += other.total
124
+
125
+
126
+ @dataclass
127
+ class ShapeStats:
128
+ """What was observed for one operation shape. No values, by construction."""
129
+
130
+ shape_id: str
131
+ group: str
132
+ entity: str
133
+ kind: str
134
+ calls: int = 0
135
+ rows: int = 0
136
+ errors: int = 0
137
+ latency: Histogram = field(default_factory=Histogram)
138
+ call_site: str | None = None
139
+
140
+ def record(self, nanoseconds: int, rows: int, failed: bool) -> None:
141
+ self.calls += 1
142
+ self.rows += rows
143
+ if failed:
144
+ self.errors += 1
145
+ self.latency.record(nanoseconds)
146
+
147
+
148
+ @dataclass
149
+ class FanOutStats:
150
+ """What was observed writing one row to one derived copy. No values, by construction.
151
+
152
+ Deliberately **not** a :class:`ShapeStats`, and that is the load-bearing decision. A fan-out is
153
+ not an operation the application asked for - it is the library keeping a copy current - so
154
+ recording it as a shape would add a write to the very counters a placement is scored on:
155
+ ``read_write_ratio`` would move because a copy exists, and a group with one copy would look
156
+ twice as write-heavy as the same group without one. The set of shape kinds that count as writes
157
+ already had four copies once (:data:`sde.shapes.WRITE_KINDS`); this is the same failure
158
+ arriving from the other side.
159
+
160
+ It is also not in :class:`GroupFeatures`, for a reason worth stating: a copy's freshness does
161
+ not score a placement. It reports the health of a copy the placement already made. Putting it
162
+ in the feature vector would bump the frozen scoring model to add an input nothing scores.
163
+ """
164
+
165
+ group: str
166
+ materialization: str
167
+ writes: int = 0
168
+ failures: int = 0
169
+ """Rows that did not reach the copy. Absence, not lateness - see :class:`CopyFreshness`."""
170
+ latency: Histogram = field(default_factory=Histogram)
171
+
172
+ def record(self, nanoseconds: int, failed: bool) -> None:
173
+ self.writes += 1
174
+ if failed:
175
+ self.failures += 1
176
+ # Recorded either way. A failed fan-out took time too, and dropping it would make the
177
+ # measured window look better precisely when the copy is in trouble.
178
+ self.latency.record(nanoseconds)
179
+
180
+
181
+ @dataclass(frozen=True)
182
+ class CopyFreshness:
183
+ """How far behind one derived copy is, measured. Requirement 5.2.
184
+
185
+ **What "behind" means here was settled by looking at the mechanism rather than at the word.**
186
+ A derived copy in this library is maintained by the fan-out in
187
+ :meth:`sde.session.Session.save` - a write in the client's own process, to the copy, straight
188
+ after the source. There is no asynchronous replication anywhere, so there is no queue to fall
189
+ behind in. Two things can therefore be true of a copy, and only two:
190
+
191
+ - it is **late** by at most the duration of that one write, which is what ``lag_p50_ms`` and
192
+ ``lag_p99_ms`` measure;
193
+ - or the write **failed**, and the row is absent rather than late. That is ``failures``, and it
194
+ is the number a lag figure would hide: a copy missing a thousand rows can have an excellent
195
+ p99.
196
+
197
+ Both are reported because they are different problems with different fixes, and a client told
198
+ only the first would read "0.9 ms behind" off a copy that is missing yesterday.
199
+
200
+ Inside a write transaction the fan-out is deferred to commit, so the window measured there
201
+ starts when the row was queued - inside the transaction - rather than at the commit. That
202
+ **overstates** the staleness by the rest of the transaction, and overstating a staleness bound
203
+ is the safe direction: the number is used to ask whether a copy is inside the budget a map
204
+ declared, and a bound that flatters is a bound nobody can rely on.
205
+ """
206
+
207
+ group: str
208
+ materialization: str
209
+ writes: int
210
+ failures: int
211
+ lag_p50_ms: float | None
212
+ lag_p99_ms: float | None
213
+
214
+ @property
215
+ def complete(self) -> bool:
216
+ """Whether every write reached the copy in this window."""
217
+ return self.failures == 0
218
+
219
+ def as_record(self) -> dict[str, Any]:
220
+ return {
221
+ "group": self.group,
222
+ "materialization": self.materialization,
223
+ "writes": self.writes,
224
+ "failures": self.failures,
225
+ "lag_p50_ms": self.lag_p50_ms,
226
+ "lag_p99_ms": self.lag_p99_ms,
227
+ "complete": self.complete,
228
+ }
229
+
230
+
231
+ @dataclass(frozen=True)
232
+ class GroupFeatures:
233
+ """The contract between telemetry and the planner.
234
+
235
+ ``None`` means *unknown*, which is not zero and is treated differently by the planner. Anything
236
+ unknown also appears in ``missing``, so a reader never has to infer absence from a null.
237
+ """
238
+
239
+ calls: int = 0
240
+ """How many operations were observed. A count, never a value.
241
+
242
+ Present because a planner comparing two groups has to know which one carries the traffic:
243
+ without it, an idle group and the group serving every request are equally important, and one
244
+ dead group can outvote the one that matters. It is also evidence about the features themselves -
245
+ a read/write ratio derived from twelve calls is arithmetic, not a measurement, and the planner
246
+ should be able to see that rather than being told a confident number.
247
+ """
248
+
249
+ read_write_ratio: float | None = None
250
+ shape_mix: Mapping[str, float] = field(default_factory=dict)
251
+ latency_p50_ms: float | None = None
252
+ latency_p99_ms: float | None = None
253
+ result_cardinality_p50: float | None = None
254
+ result_cardinality_p99: float | None = None
255
+ total_bytes: int | None = None
256
+ daily_growth_bytes: int | None = None
257
+ index_to_table_ratio: float | None = None
258
+ pk_access_share: float | None = None
259
+ has_time_dimension: bool = False
260
+ time_filtered_share: float | None = None
261
+ distinct_shapes: int = 0
262
+ write_burstiness: float | None = None
263
+ error_share: float | None = None
264
+ missing: frozenset[str] = frozenset()
265
+ complete: bool = True
266
+
267
+ def as_record(self) -> dict[str, Any]:
268
+ """The feature vector as the document that crosses the boundary to the control plane.
269
+
270
+ **A field with no value is omitted, and ``missing`` is what says so.** Emitting a null
271
+ would work too - the reader treats absent and null alike - but omitting is the honest
272
+ spelling of "not measured", and it keeps this method from having an opinion about what a
273
+ null means.
274
+
275
+ The field list is **derived** from this dataclass rather than written out. That is
276
+ requirement 6.7 from the writing side: a field added to the library joins this document
277
+ instead of being silently dropped, which is the mirror of the defect the control plane's
278
+ reader had - a hand-written list, against a dataclass where every field has a default, so
279
+ the next field would have been recorded as measured.
280
+
281
+ No float is formatted here and none is rounded. Every number in this document is either a
282
+ ratio of two integers or a bucket edge divided by a million, so two languages compute the
283
+ same IEEE 754 double - see :meth:`Window.as_record` for why that matters and where the
284
+ limit of the claim is.
285
+ """
286
+ body: dict[str, Any] = {}
287
+ for name in MEASURED_FIELDS:
288
+ value = getattr(self, name)
289
+ if value is None:
290
+ continue
291
+ body[name] = dict(value) if isinstance(value, Mapping) else value
292
+ body["missing"] = sorted(self.missing)
293
+ body["complete"] = self.complete
294
+ return body
295
+
296
+
297
+ MEASURED_FIELDS: tuple[str, ...] = tuple(
298
+ spec.name
299
+ for spec in dataclass_fields(GroupFeatures)
300
+ if spec.name not in ("missing", "complete")
301
+ )
302
+ """Every field of :class:`GroupFeatures` that carries a measurement.
303
+
304
+ Derived from the dataclass, and the two names excluded are bookkeeping *about* the measurement
305
+ rather than part of it. Both this module and the control plane's reader work from this same source,
306
+ so a field added to the feature vector cannot be written by one side and ignored by the other.
307
+ """
308
+
309
+
310
+ @dataclass(frozen=True)
311
+ class Window:
312
+ """One aggregation period, ready to send.
313
+
314
+ ``complete`` is false when the library could not reach the control plane for part of the period,
315
+ or when the buffer dropped windows. A window that is not complete is still sent - the planner
316
+ needs to know traffic existed - but it may not be used to justify a migration.
317
+ """
318
+
319
+ model_version: str
320
+ started_ns: int
321
+ ended_ns: int
322
+ shapes: Sequence[ShapeStats]
323
+ complete: bool = True
324
+ dropped_windows: int = 0
325
+ fanned: Sequence[FanOutStats] = ()
326
+ """What the fan-out to each derived copy did. Empty when the group has no copy, which is the
327
+ ordinary case - a copy exists during a migration and while a derived materialisation is in the
328
+ map, not otherwise."""
329
+
330
+ def copies(self, group: str) -> tuple[CopyFreshness, ...]:
331
+ """How far behind each of this group's derived copies ran, sorted by materialisation.
332
+
333
+ Read out of the histogram rather than stored, like every other percentile here. A stored
334
+ percentile is a second copy of a fact that changes when the samples do.
335
+ """
336
+ return tuple(
337
+ CopyFreshness(
338
+ group=stats.group,
339
+ materialization=stats.materialization,
340
+ writes=stats.writes,
341
+ failures=stats.failures,
342
+ lag_p50_ms=stats.latency.percentile_ms(0.50),
343
+ lag_p99_ms=stats.latency.percentile_ms(0.99),
344
+ )
345
+ for stats in sorted(
346
+ (s for s in self.fanned if s.group == group),
347
+ key=lambda s: s.materialization,
348
+ )
349
+ )
350
+
351
+ def features(self, group: str, *, has_time_dimension: bool = False) -> GroupFeatures:
352
+ """Fold this window's records for one group into the planner's feature vector."""
353
+ records = [s for s in self.shapes if s.group == group]
354
+ if not records:
355
+ # `no_traffic` is the *reason*, and the unknown fields are named too - by the same
356
+ # derivation as the branch below, because two branches of one function computing
357
+ # `missing` by two different rules is the defect this set exists to prevent. An empty
358
+ # window really did not measure a read/write ratio, and saying only "no traffic" leaves
359
+ # a reader to work out which fields that implies.
360
+ return _with_missing(
361
+ GroupFeatures(complete=self.complete, distinct_shapes=0), also={"no_traffic"}
362
+ )
363
+
364
+ writes = sum(s.calls for s in records if s.kind in WRITE_KINDS)
365
+ reads = sum(s.calls for s in records if s.kind not in WRITE_KINDS)
366
+ calls = writes + reads
367
+
368
+ latency = Histogram()
369
+ for record in records:
370
+ latency.merge(record.latency)
371
+
372
+ mix: dict[str, float] = {}
373
+ for record in records:
374
+ mix[record.kind] = mix.get(record.kind, 0.0) + record.calls
375
+ mix = {kind: hits / calls for kind, hits in sorted(mix.items())} if calls else {}
376
+
377
+ # A failed call counts in the latency histogram and **not** here, and the two answers have
378
+ # different reasons rather than one convention. A failure took time, so dropping it from
379
+ # the histogram would flatter a window precisely when the engine is in trouble. It returned
380
+ # no rows because it failed rather than because the data is sparse, so averaging that zero
381
+ # in understates how many rows a read of this shape returns - and `error_share` already
382
+ # carries the failure rate, so smearing it into a second feature is one fact in two places.
383
+ # A shape whose every call failed contributes nothing rather than a zero. `telemetry/008`.
384
+ read_records = [
385
+ s for s in records if s.kind not in WRITE_KINDS and s.calls > s.errors
386
+ ]
387
+ cardinalities = sorted(s.rows / (s.calls - s.errors) for s in read_records)
388
+
389
+ pk_calls = sum(s.calls for s in records if s.kind == "point_read")
390
+ errors = sum(s.errors for s in records)
391
+
392
+ measured = GroupFeatures(
393
+ calls=calls,
394
+ read_write_ratio=(reads / writes) if writes else None,
395
+ shape_mix=mix,
396
+ latency_p50_ms=latency.percentile_ms(0.5),
397
+ latency_p99_ms=latency.percentile_ms(0.99),
398
+ result_cardinality_p50=_at(cardinalities, 0.5),
399
+ result_cardinality_p99=_at(cardinalities, 0.99),
400
+ pk_access_share=(pk_calls / calls) if calls else None,
401
+ has_time_dimension=has_time_dimension,
402
+ distinct_shapes=len(records),
403
+ error_share=(errors / calls) if calls else None,
404
+ complete=self.complete,
405
+ )
406
+ return _with_missing(measured)
407
+
408
+
409
+
410
+
411
+ def as_record(self, model: LogicalModel) -> dict[str, Any]:
412
+ """This window as the document the control plane reads. Numbers, never rows.
413
+
414
+ **The artefact this library existed without.** Everything above measured traffic and
415
+ nothing turned a window into the file we are handed - so the walkthrough's step 8, "the
416
+ library measures traffic and hands us a window", was a literal dictionary typed into the
417
+ example, and every client in every language would have written their own. Two clients of
418
+ one model would then produce two documents from identical traffic, and the planner would
419
+ score whichever one it was given. Same shape of gap as the two before it: both halves
420
+ worked and nothing joined them.
421
+
422
+ The model is required and it is checked. Only one fact is read from it - whether a group
423
+ carries a time dimension, which is decided by declared *type* and never by a field's name -
424
+ but a window serialised against the wrong model would attach that fact to the wrong groups
425
+ and claim `has_time_dimension: false` for a group that has one. False is a claim; a
426
+ measurement this library cannot make has to be absent, which is what everything else in
427
+ here is careful about.
428
+
429
+ **This document is deliberately not canonical, and that needs saying because §1 of the
430
+ format contract rejects floating point outright.** Its reason is that a float's textual
431
+ form differs between languages, and almost every number here is a float. This is not
432
+ signed, not hashed and never compared for equality, so the rule it breaks does not apply -
433
+ but the thing that makes the family checkable at all is narrower and worth stating: every
434
+ number in this document is either **a ratio of two integers** or **a bucket edge divided by
435
+ a million**, and IEEE 754 requires division to be correctly rounded. So two languages
436
+ compute the same double from the same traffic even where they would print it differently,
437
+ and the `telemetry/` vectors compare numbers rather than bytes for exactly that reason.
438
+ """
439
+ if model.version != self.model_version:
440
+ raise ValueError(
441
+ f"this window measured model version {self.model_version} and it is being "
442
+ f"serialised against {model.version}. Only one fact is read from the model here - "
443
+ f"whether a group has a time dimension - and reading it from another model would "
444
+ f"attach it to the wrong groups. Keep the model the recorder was created for, or "
445
+ f"drop the window: a window measured against a model that no longer exists cannot "
446
+ f"be scored against the one that does."
447
+ )
448
+ by_name = {group.name: group for group in colocation_groups(model)}
449
+ named = sorted({stats.group for stats in self.shapes} | {s.group for s in self.fanned})
450
+ unknown = [name for name in named if name not in by_name]
451
+ if unknown:
452
+ raise ValueError(
453
+ f"this window has records for groups this model does not have: {unknown}. The "
454
+ f"recorder is given a model version and the groups come from the operations it "
455
+ f"observed, so this is a session routing a model other than the one measured."
456
+ )
457
+
458
+ groups: dict[str, Any] = {}
459
+ for name in named:
460
+ record = self.features(
461
+ name, has_time_dimension=has_time_dimension(model, by_name[name])
462
+ ).as_record()
463
+ copies = [copy.as_record() for copy in self.copies(name)]
464
+ if copies:
465
+ # Absent rather than empty when the group has no derived copy, for the reason
466
+ # `also_write` is absent rather than empty in a map: absent says "not doing this"
467
+ # and an empty list says "considered and found none", which is a stronger claim.
468
+ record["copies"] = copies
469
+ groups[name] = record
470
+
471
+ return {
472
+ "model_version": self.model_version,
473
+ "complete": self.complete,
474
+ "dropped_windows": self.dropped_windows,
475
+ "groups": groups,
476
+ }
477
+
478
+
479
+ def _with_missing(
480
+ measured: GroupFeatures, *, also: Collection[str] = ()
481
+ ) -> GroupFeatures:
482
+ """The same features with ``missing`` **derived** from which values came out unknown.
483
+
484
+ Derived rather than listed alongside them, so the two cannot disagree - and they did.
485
+ ``time_filtered_share`` is a field this library never measures and it was left null and absent
486
+ from this set, which breaks the one promise the set makes: that a reader never has to infer
487
+ absence from a null. Four other fields were named by hand and would have stayed the only ones.
488
+
489
+ What is unknown and why, because a derivation hides the reasons. Engine-side sizes
490
+ (``total_bytes``, ``index_to_table_ratio``) need a catalogue read, which is an adapter
491
+ capability rather than a measurement. ``daily_growth_bytes`` and ``write_burstiness`` need two
492
+ samples over time and a window is one. ``time_filtered_share`` would need the *arguments* of a
493
+ call, and this library records shapes and never values. Zero bytes and unknown bytes lead a
494
+ planner to opposite conclusions, which is the whole reason for naming them rather than
495
+ defaulting them.
496
+
497
+ ``also`` carries reasons that are not field names - ``no_traffic`` is the only one - because a
498
+ reader of the window needs to know the difference between "this group was idle" and "this
499
+ field is not measurable".
500
+ """
501
+ return replace(
502
+ measured,
503
+ missing=frozenset(also)
504
+ | frozenset(name for name in MEASURED_FIELDS if getattr(measured, name) is None),
505
+ )
506
+
507
+
508
+ def _at(ordered: list[float], fraction: float) -> float | None:
509
+ """Nearest-rank: the smallest sample at least ``fraction`` of the data is not above.
510
+
511
+ **The same rank rule as** :meth:`Histogram.percentile_ms`, and it did not used to be. This
512
+ function selected ``ordered[int(len * fraction)]`` - the floor - while the histogram takes the
513
+ first bucket whose cumulative count reaches ``fraction * count``, which is the ceiling. The two
514
+ agree on every sample set with an odd count, and every case in ``telemetry/`` had one, so one
515
+ window document carried two percentile conventions and nothing could see it: a third
516
+ implementation swapped one for the other and the whole family stayed green.
517
+
518
+ They differ exactly when ``fraction * len`` is an integer. Two samples at p50 is that case:
519
+ ``[3.0, 300.0]`` reported 300 here and the *lower* bucket in the histogram, for the same
520
+ reason applied twice in opposite directions. ``telemetry/007`` pins it.
521
+ """
522
+ if not ordered:
523
+ return None
524
+ rank = max(1, math.ceil(fraction * len(ordered))) - 1
525
+ return ordered[min(len(ordered) - 1, rank)]
526
+
527
+
528
+ # Types that make a field a time dimension. Recognised by **type**, never by name - see below.
529
+ _TIME_TYPES = frozenset({"date", "timestamp", "timestamptz"})
530
+
531
+
532
+ def has_time_dimension(model: LogicalModel, group: Group) -> bool:
533
+ """Does any entity in this group carry a time dimension?
534
+
535
+ Decided by the declared type and never by the field's name. Two reasons, and the second is the
536
+ one that made it a rule rather than a preference.
537
+
538
+ A name is not evidence: `created_at` typed as a string is a string, and treating it as a
539
+ timestamp would have the planner recommend time partitioning on a column no engine can
540
+ range-scan usefully.
541
+
542
+ And a client may hash identifier names (requirement 11.3) so that we never see them. A
543
+ derivation that reads names would silently produce different answers with hashing on and off -
544
+ which is exactly the property the conformance vector for hashed models exists to forbid.
545
+ Deciding by type also means models written in other natural languages work identically, which is
546
+ a free consequence of getting this right for the other reason.
547
+ """
548
+ for member in group.members:
549
+ spec = model.entity(member)
550
+ for spec_field in spec.fields:
551
+ if spec_field.type in _TIME_TYPES:
552
+ return True
553
+ return False
554
+
555
+
556
+ class Recorder:
557
+ """Accumulates records, rolls windows, and drops telemetry rather than anything else.
558
+
559
+ Thread safety is a single lock taken only when a shape is first seen or a window rolls - not on
560
+ every record. Counters for an existing shape are updated without it: two threads racing on the
561
+ same shape can lose a call from the count, and that is the right trade. Telemetry informs a
562
+ placement decision made over days; a lock on the hot path would cost every operation forever.
563
+ """
564
+
565
+ def __init__(self, model_version: str, *, max_windows: int = 64) -> None:
566
+ self._model_version = model_version
567
+ self._lock = threading.Lock()
568
+ self._current: dict[str, ShapeStats] = {}
569
+ self._fanned: dict[tuple[str, str], FanOutStats] = {}
570
+ self._started_ns = _now()
571
+ self._windows: deque[Window] = deque(maxlen=max_windows)
572
+ self._dropped = 0
573
+ self._incomplete = False
574
+
575
+ # --- recording ---------------------------------------------------------------------------
576
+
577
+ def record(
578
+ self,
579
+ *,
580
+ shape_id: str,
581
+ group: str,
582
+ entity: str,
583
+ kind: str,
584
+ nanoseconds: int,
585
+ rows: int = 0,
586
+ failed: bool = False,
587
+ ) -> None:
588
+ """Record one operation. Never raises, never blocks on the common path."""
589
+ guard(
590
+ "telemetry.record",
591
+ lambda: self._record(shape_id, group, entity, kind, nanoseconds, rows, failed),
592
+ )
593
+
594
+ def _record(
595
+ self,
596
+ shape_id: str,
597
+ group: str,
598
+ entity: str,
599
+ kind: str,
600
+ nanoseconds: int,
601
+ rows: int,
602
+ failed: bool,
603
+ ) -> None:
604
+ stats = self._current.get(shape_id)
605
+ if stats is None:
606
+ with self._lock:
607
+ stats = self._current.get(shape_id)
608
+ if stats is None:
609
+ stats = ShapeStats(
610
+ shape_id=shape_id, group=group, entity=entity, kind=kind,
611
+ call_site=_call_site(),
612
+ )
613
+ self._current[shape_id] = stats
614
+ stats.record(nanoseconds, rows, failed)
615
+
616
+ def record_fan_out(
617
+ self, *, group: str, materialization: str, nanoseconds: int, failed: bool = False
618
+ ) -> None:
619
+ """Record one write to one derived copy. Never raises, never blocks on the common path.
620
+
621
+ A separate entry point from :meth:`record` rather than a shape kind, because a fan-out is
622
+ not an operation the application asked for - see :class:`FanOutStats`.
623
+ """
624
+ guard(
625
+ "telemetry.record_fan_out",
626
+ lambda: self._record_fan_out(group, materialization, nanoseconds, failed),
627
+ )
628
+
629
+ def _record_fan_out(
630
+ self, group: str, materialization: str, nanoseconds: int, failed: bool
631
+ ) -> None:
632
+ key = (group, materialization)
633
+ stats = self._fanned.get(key)
634
+ if stats is None:
635
+ with self._lock:
636
+ stats = self._fanned.get(key)
637
+ if stats is None:
638
+ stats = FanOutStats(group=group, materialization=materialization)
639
+ self._fanned[key] = stats
640
+ stats.record(nanoseconds, failed)
641
+
642
+ # --- windows -----------------------------------------------------------------------------
643
+
644
+ def roll(self) -> Window | None:
645
+ """Close the current period and queue it.
646
+
647
+ Returns the window, or None if nothing was recorded in it.
648
+ """
649
+ return guard("telemetry.roll", self._roll)
650
+
651
+ def _roll(self) -> Window | None:
652
+ with self._lock:
653
+ if not self._current and not self._fanned:
654
+ # `not self._current` alone would drop a window holding only fan-out records. That
655
+ # cannot arise from an application write - a write records a shape and then fans
656
+ # out - but it can from a backfill replaying rows, and a window silently discarded
657
+ # is the shape of gap that makes a client's copy look healthier than it is.
658
+ return None
659
+ window = Window(
660
+ model_version=self._model_version,
661
+ started_ns=self._started_ns,
662
+ ended_ns=_now(),
663
+ shapes=tuple(self._current.values()),
664
+ complete=not self._incomplete,
665
+ dropped_windows=self._dropped,
666
+ fanned=tuple(self._fanned.values()),
667
+ )
668
+ self._current = {}
669
+ self._fanned = {}
670
+ self._started_ns = _now()
671
+ self._incomplete = False
672
+
673
+ # A full buffer drops the oldest window and says so in the next one. Telemetry is the
674
+ # thing that gets lost when we run out of room - never a write, never an operation.
675
+ if len(self._windows) == self._windows.maxlen:
676
+ self._dropped += 1
677
+ log("sde.telemetry.dropped", dropped=self._dropped)
678
+ self._windows.append(window)
679
+ log(
680
+ "sde.telemetry.window_closed",
681
+ model_version=window.model_version,
682
+ shapes=len(window.shapes),
683
+ complete=window.complete,
684
+ )
685
+ return window
686
+
687
+ def pending(self) -> tuple[Window, ...]:
688
+ with self._lock:
689
+ return tuple(self._windows)
690
+
691
+ def mark_incomplete(self) -> None:
692
+ """Called by the application when part of the current period was not recorded.
693
+
694
+ The flag travels to us in :attr:`Window.complete` and the planner reads it, because a
695
+ window missing a slice of the traffic must not be scored as though it were the whole
696
+ period: an application that was down for six of the last twenty-four hours looks
697
+ idle-and-cheap on figures nobody marked as partial.
698
+
699
+ Nothing here decides when that happened. The application collects these windows and hands
700
+ them to us; only it knows whether a collection was skipped.
701
+ """
702
+ self._incomplete = True
703
+
704
+ def acknowledge(self, count: int) -> None:
705
+ """Drop the oldest ``count`` windows once the application has taken them.
706
+
707
+ The delivering party is the client's own process, and there is no other. This library
708
+ never opens a connection to us - the buffer is read with :meth:`pending`, written wherever
709
+ the application writes it, and that file is what we are handed.
710
+ """
711
+ with self._lock:
712
+ for _ in range(min(count, len(self._windows))):
713
+ self._windows.popleft()
714
+
715
+
716
+ def _now() -> int:
717
+ return int(__import__("time").perf_counter_ns())
718
+
719
+
720
+ def _call_site() -> str | None:
721
+ """Where in the client's code this shape is used. Walked once per shape, never per call.
722
+
723
+ Best effort by design: a frame walk is not free and a missing call site costs a little
724
+ diagnostic value, while a frame walk on every operation would cost the latency budget.
725
+ """
726
+
727
+ def walk() -> str | None:
728
+ frame: Any = sys._getframe(1)
729
+ while frame is not None:
730
+ name = frame.f_globals.get("__name__", "")
731
+ if not name.startswith("sde"):
732
+ return f"{frame.f_code.co_filename}:{frame.f_lineno}"
733
+ frame = frame.f_back
734
+ return None
735
+
736
+ return guard("telemetry.call_site", walk)