smart-data-engine-sdk 0.1.0.dev0__py3-none-any.whl
This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
- sde/__init__.py +226 -0
- sde/canonical.py +141 -0
- sde/capabilities.py +62 -0
- sde/engines/__init__.py +0 -0
- sde/engines/clickhouse.py +689 -0
- sde/engines/orderbook.py +454 -0
- sde/engines/postgres.py +672 -0
- sde/entity.py +170 -0
- sde/errors.py +88 -0
- sde/explain.py +300 -0
- sde/groups.py +97 -0
- sde/hashing.py +242 -0
- sde/infer.py +461 -0
- sde/internal.py +90 -0
- sde/layout.py +660 -0
- sde/logging.py +132 -0
- sde/migration.py +820 -0
- sde/model.py +482 -0
- sde/placement.py +818 -0
- sde/py.typed +0 -0
- sde/routing.py +85 -0
- sde/schema.py +370 -0
- sde/session.py +507 -0
- sde/shapes.py +153 -0
- sde/telemetry.py +736 -0
- sde/testing/__init__.py +14 -0
- sde/testing/loader.py +175 -0
- sde/testing/memory.py +318 -0
- sde/types.py +228 -0
- sde/watermark.py +222 -0
- smart_data_engine_sdk-0.1.0.dev0.dist-info/METADATA +152 -0
- smart_data_engine_sdk-0.1.0.dev0.dist-info/RECORD +35 -0
- smart_data_engine_sdk-0.1.0.dev0.dist-info/WHEEL +4 -0
- smart_data_engine_sdk-0.1.0.dev0.dist-info/licenses/LICENSE +201 -0
- smart_data_engine_sdk-0.1.0.dev0.dist-info/licenses/NOTICE +13 -0
sde/telemetry.py
ADDED
|
@@ -0,0 +1,736 @@
|
|
|
1
|
+
"""Measuring what the application actually does, without ever seeing what it does it to.
|
|
2
|
+
|
|
3
|
+
This is the input the placement decision is made from, so its shape matters more than its precision.
|
|
4
|
+
Three constraints shaped everything here.
|
|
5
|
+
|
|
6
|
+
**It carries no values.** A record is keyed by an operation shape, which is assembled from the
|
|
7
|
+
structure of a call and never sees its arguments. There is no code path by which a customer's row
|
|
8
|
+
reaches a telemetry record, which is why this file can be read by a client and believed.
|
|
9
|
+
|
|
10
|
+
**It cannot cost anything.** Routing already has a one percent budget for the whole library, and
|
|
11
|
+
recording happens on the same path. So: no locks on the hot path per record, no string formatting,
|
|
12
|
+
no stack walking except once per shape, and a histogram rather than a list of samples.
|
|
13
|
+
|
|
14
|
+
**It cannot fail the caller.** Every entry point is wrapped in :func:`~sde.internal.guard`. A bug in
|
|
15
|
+
an aggregation counter must not take down somebody's request - and the failure is counted, so it is
|
|
16
|
+
not invisible either.
|
|
17
|
+
|
|
18
|
+
The histogram deserves a word, because it is the one deliberate loss of precision. Latency lands in
|
|
19
|
+
exponential buckets, so a percentile read out of it is approximate - within one bucket width, which
|
|
20
|
+
is a factor of two at the extremes. That is ample for the decision it feeds: the planner cares
|
|
21
|
+
whether a group's reads are microseconds or milliseconds, not whether p99 is 412 or 431
|
|
22
|
+
microseconds. Keeping exact samples would mean either unbounded memory or reservoir sampling, and
|
|
23
|
+
reservoir sampling gets the tail wrong in exactly the region the planner looks at.
|
|
24
|
+
"""
|
|
25
|
+
|
|
26
|
+
from __future__ import annotations
|
|
27
|
+
|
|
28
|
+
import math
|
|
29
|
+
import sys
|
|
30
|
+
import threading
|
|
31
|
+
from collections import deque
|
|
32
|
+
from collections.abc import Collection, Mapping, Sequence
|
|
33
|
+
from dataclasses import dataclass, field, replace
|
|
34
|
+
from dataclasses import fields as dataclass_fields
|
|
35
|
+
from typing import Any
|
|
36
|
+
|
|
37
|
+
from .groups import Group, colocation_groups
|
|
38
|
+
from .internal import guard
|
|
39
|
+
from .logging import log
|
|
40
|
+
from .model import LogicalModel
|
|
41
|
+
from .shapes import WRITE_KINDS
|
|
42
|
+
|
|
43
|
+
__all__ = [
|
|
44
|
+
"MEASURED_FIELDS",
|
|
45
|
+
"CopyFreshness",
|
|
46
|
+
"FanOutStats",
|
|
47
|
+
"GroupFeatures",
|
|
48
|
+
"Histogram",
|
|
49
|
+
"Recorder",
|
|
50
|
+
"ShapeStats",
|
|
51
|
+
"Window",
|
|
52
|
+
"has_time_dimension",
|
|
53
|
+
]
|
|
54
|
+
|
|
55
|
+
# 1 µs to about 17 s, doubling. Twenty-five buckets is enough to tell a cache hit from a full scan,
|
|
56
|
+
# which is the distinction the planner actually acts on.
|
|
57
|
+
BUCKET_COUNT = 25
|
|
58
|
+
BUCKET_BASE_NS = 1_000
|
|
59
|
+
|
|
60
|
+
|
|
61
|
+
class Histogram:
|
|
62
|
+
"""Exponential-bucket histogram. Fixed memory, O(1) record, approximate percentiles."""
|
|
63
|
+
|
|
64
|
+
__slots__ = ("buckets", "count", "total")
|
|
65
|
+
|
|
66
|
+
def __init__(self) -> None:
|
|
67
|
+
self.buckets = [0] * BUCKET_COUNT
|
|
68
|
+
self.count = 0
|
|
69
|
+
self.total = 0
|
|
70
|
+
|
|
71
|
+
def record(self, nanoseconds: int) -> None:
|
|
72
|
+
"""Put one duration in its bucket. Integer arithmetic only, and that is the point.
|
|
73
|
+
|
|
74
|
+
This used to read ``int(math.log2(nanoseconds / BUCKET_BASE_NS)) + 1``, which is the same
|
|
75
|
+
function and the wrong way to compute it once a second language has to agree. ``log2`` is
|
|
76
|
+
not required by IEEE 754 to be correctly rounded, so two libm implementations may differ in
|
|
77
|
+
the last bit - and one bit at a power-of-two boundary is a different bucket, which is a
|
|
78
|
+
different p99 for identical traffic. The bit length of the integer quotient is exact
|
|
79
|
+
everywhere, and it is cheaper on a path that runs per operation.
|
|
80
|
+
|
|
81
|
+
Verified rather than asserted: the two forms were compared over every value below 40 000,
|
|
82
|
+
every power-of-two boundary and its neighbours, and 400 000 random durations up to 10^13 ns.
|
|
83
|
+
Zero disagreements.
|
|
84
|
+
|
|
85
|
+
**And the vectors cannot see the difference, which is the point of saying so here.** An
|
|
86
|
+
earlier version of this note claimed ``telemetry/002`` pins it. Measured: replacing this
|
|
87
|
+
with the logarithm form passes every vector, because glibc's ``log2`` and V8's are both
|
|
88
|
+
exact at a power of two - so the two runtimes we have agree, and the hazard is a *third*
|
|
89
|
+
libm that does not. A property no output can distinguish on the machines available is not
|
|
90
|
+
one a vector can hold, so it is held statically instead: see
|
|
91
|
+
``test_the_bucket_index_never_reaches_for_a_logarithm``.
|
|
92
|
+
"""
|
|
93
|
+
self.count += 1
|
|
94
|
+
self.total += nanoseconds
|
|
95
|
+
if nanoseconds < BUCKET_BASE_NS:
|
|
96
|
+
self.buckets[0] += 1
|
|
97
|
+
return
|
|
98
|
+
index: int = min(BUCKET_COUNT - 1, (nanoseconds // BUCKET_BASE_NS).bit_length())
|
|
99
|
+
self.buckets[index] += 1
|
|
100
|
+
|
|
101
|
+
def percentile_ms(self, fraction: float) -> float | None:
|
|
102
|
+
"""Approximate percentile in milliseconds, or None if nothing was recorded.
|
|
103
|
+
|
|
104
|
+
Returns the *upper* edge of the bucket the percentile falls in. Rounding up rather than
|
|
105
|
+
interpolating is deliberate: a placement decision made on an optimistic latency figure is
|
|
106
|
+
the wrong kind of wrong.
|
|
107
|
+
"""
|
|
108
|
+
if self.count == 0:
|
|
109
|
+
return None
|
|
110
|
+
target = fraction * self.count
|
|
111
|
+
seen = 0
|
|
112
|
+
for index, hits in enumerate(self.buckets):
|
|
113
|
+
seen += hits
|
|
114
|
+
if seen >= target:
|
|
115
|
+
upper_ns: int = BUCKET_BASE_NS * (2**index)
|
|
116
|
+
return upper_ns / 1_000_000
|
|
117
|
+
return None
|
|
118
|
+
|
|
119
|
+
def merge(self, other: Histogram) -> None:
|
|
120
|
+
for index, hits in enumerate(other.buckets):
|
|
121
|
+
self.buckets[index] += hits
|
|
122
|
+
self.count += other.count
|
|
123
|
+
self.total += other.total
|
|
124
|
+
|
|
125
|
+
|
|
126
|
+
@dataclass
|
|
127
|
+
class ShapeStats:
|
|
128
|
+
"""What was observed for one operation shape. No values, by construction."""
|
|
129
|
+
|
|
130
|
+
shape_id: str
|
|
131
|
+
group: str
|
|
132
|
+
entity: str
|
|
133
|
+
kind: str
|
|
134
|
+
calls: int = 0
|
|
135
|
+
rows: int = 0
|
|
136
|
+
errors: int = 0
|
|
137
|
+
latency: Histogram = field(default_factory=Histogram)
|
|
138
|
+
call_site: str | None = None
|
|
139
|
+
|
|
140
|
+
def record(self, nanoseconds: int, rows: int, failed: bool) -> None:
|
|
141
|
+
self.calls += 1
|
|
142
|
+
self.rows += rows
|
|
143
|
+
if failed:
|
|
144
|
+
self.errors += 1
|
|
145
|
+
self.latency.record(nanoseconds)
|
|
146
|
+
|
|
147
|
+
|
|
148
|
+
@dataclass
|
|
149
|
+
class FanOutStats:
|
|
150
|
+
"""What was observed writing one row to one derived copy. No values, by construction.
|
|
151
|
+
|
|
152
|
+
Deliberately **not** a :class:`ShapeStats`, and that is the load-bearing decision. A fan-out is
|
|
153
|
+
not an operation the application asked for - it is the library keeping a copy current - so
|
|
154
|
+
recording it as a shape would add a write to the very counters a placement is scored on:
|
|
155
|
+
``read_write_ratio`` would move because a copy exists, and a group with one copy would look
|
|
156
|
+
twice as write-heavy as the same group without one. The set of shape kinds that count as writes
|
|
157
|
+
already had four copies once (:data:`sde.shapes.WRITE_KINDS`); this is the same failure
|
|
158
|
+
arriving from the other side.
|
|
159
|
+
|
|
160
|
+
It is also not in :class:`GroupFeatures`, for a reason worth stating: a copy's freshness does
|
|
161
|
+
not score a placement. It reports the health of a copy the placement already made. Putting it
|
|
162
|
+
in the feature vector would bump the frozen scoring model to add an input nothing scores.
|
|
163
|
+
"""
|
|
164
|
+
|
|
165
|
+
group: str
|
|
166
|
+
materialization: str
|
|
167
|
+
writes: int = 0
|
|
168
|
+
failures: int = 0
|
|
169
|
+
"""Rows that did not reach the copy. Absence, not lateness - see :class:`CopyFreshness`."""
|
|
170
|
+
latency: Histogram = field(default_factory=Histogram)
|
|
171
|
+
|
|
172
|
+
def record(self, nanoseconds: int, failed: bool) -> None:
|
|
173
|
+
self.writes += 1
|
|
174
|
+
if failed:
|
|
175
|
+
self.failures += 1
|
|
176
|
+
# Recorded either way. A failed fan-out took time too, and dropping it would make the
|
|
177
|
+
# measured window look better precisely when the copy is in trouble.
|
|
178
|
+
self.latency.record(nanoseconds)
|
|
179
|
+
|
|
180
|
+
|
|
181
|
+
@dataclass(frozen=True)
|
|
182
|
+
class CopyFreshness:
|
|
183
|
+
"""How far behind one derived copy is, measured. Requirement 5.2.
|
|
184
|
+
|
|
185
|
+
**What "behind" means here was settled by looking at the mechanism rather than at the word.**
|
|
186
|
+
A derived copy in this library is maintained by the fan-out in
|
|
187
|
+
:meth:`sde.session.Session.save` - a write in the client's own process, to the copy, straight
|
|
188
|
+
after the source. There is no asynchronous replication anywhere, so there is no queue to fall
|
|
189
|
+
behind in. Two things can therefore be true of a copy, and only two:
|
|
190
|
+
|
|
191
|
+
- it is **late** by at most the duration of that one write, which is what ``lag_p50_ms`` and
|
|
192
|
+
``lag_p99_ms`` measure;
|
|
193
|
+
- or the write **failed**, and the row is absent rather than late. That is ``failures``, and it
|
|
194
|
+
is the number a lag figure would hide: a copy missing a thousand rows can have an excellent
|
|
195
|
+
p99.
|
|
196
|
+
|
|
197
|
+
Both are reported because they are different problems with different fixes, and a client told
|
|
198
|
+
only the first would read "0.9 ms behind" off a copy that is missing yesterday.
|
|
199
|
+
|
|
200
|
+
Inside a write transaction the fan-out is deferred to commit, so the window measured there
|
|
201
|
+
starts when the row was queued - inside the transaction - rather than at the commit. That
|
|
202
|
+
**overstates** the staleness by the rest of the transaction, and overstating a staleness bound
|
|
203
|
+
is the safe direction: the number is used to ask whether a copy is inside the budget a map
|
|
204
|
+
declared, and a bound that flatters is a bound nobody can rely on.
|
|
205
|
+
"""
|
|
206
|
+
|
|
207
|
+
group: str
|
|
208
|
+
materialization: str
|
|
209
|
+
writes: int
|
|
210
|
+
failures: int
|
|
211
|
+
lag_p50_ms: float | None
|
|
212
|
+
lag_p99_ms: float | None
|
|
213
|
+
|
|
214
|
+
@property
|
|
215
|
+
def complete(self) -> bool:
|
|
216
|
+
"""Whether every write reached the copy in this window."""
|
|
217
|
+
return self.failures == 0
|
|
218
|
+
|
|
219
|
+
def as_record(self) -> dict[str, Any]:
|
|
220
|
+
return {
|
|
221
|
+
"group": self.group,
|
|
222
|
+
"materialization": self.materialization,
|
|
223
|
+
"writes": self.writes,
|
|
224
|
+
"failures": self.failures,
|
|
225
|
+
"lag_p50_ms": self.lag_p50_ms,
|
|
226
|
+
"lag_p99_ms": self.lag_p99_ms,
|
|
227
|
+
"complete": self.complete,
|
|
228
|
+
}
|
|
229
|
+
|
|
230
|
+
|
|
231
|
+
@dataclass(frozen=True)
|
|
232
|
+
class GroupFeatures:
|
|
233
|
+
"""The contract between telemetry and the planner.
|
|
234
|
+
|
|
235
|
+
``None`` means *unknown*, which is not zero and is treated differently by the planner. Anything
|
|
236
|
+
unknown also appears in ``missing``, so a reader never has to infer absence from a null.
|
|
237
|
+
"""
|
|
238
|
+
|
|
239
|
+
calls: int = 0
|
|
240
|
+
"""How many operations were observed. A count, never a value.
|
|
241
|
+
|
|
242
|
+
Present because a planner comparing two groups has to know which one carries the traffic:
|
|
243
|
+
without it, an idle group and the group serving every request are equally important, and one
|
|
244
|
+
dead group can outvote the one that matters. It is also evidence about the features themselves -
|
|
245
|
+
a read/write ratio derived from twelve calls is arithmetic, not a measurement, and the planner
|
|
246
|
+
should be able to see that rather than being told a confident number.
|
|
247
|
+
"""
|
|
248
|
+
|
|
249
|
+
read_write_ratio: float | None = None
|
|
250
|
+
shape_mix: Mapping[str, float] = field(default_factory=dict)
|
|
251
|
+
latency_p50_ms: float | None = None
|
|
252
|
+
latency_p99_ms: float | None = None
|
|
253
|
+
result_cardinality_p50: float | None = None
|
|
254
|
+
result_cardinality_p99: float | None = None
|
|
255
|
+
total_bytes: int | None = None
|
|
256
|
+
daily_growth_bytes: int | None = None
|
|
257
|
+
index_to_table_ratio: float | None = None
|
|
258
|
+
pk_access_share: float | None = None
|
|
259
|
+
has_time_dimension: bool = False
|
|
260
|
+
time_filtered_share: float | None = None
|
|
261
|
+
distinct_shapes: int = 0
|
|
262
|
+
write_burstiness: float | None = None
|
|
263
|
+
error_share: float | None = None
|
|
264
|
+
missing: frozenset[str] = frozenset()
|
|
265
|
+
complete: bool = True
|
|
266
|
+
|
|
267
|
+
def as_record(self) -> dict[str, Any]:
|
|
268
|
+
"""The feature vector as the document that crosses the boundary to the control plane.
|
|
269
|
+
|
|
270
|
+
**A field with no value is omitted, and ``missing`` is what says so.** Emitting a null
|
|
271
|
+
would work too - the reader treats absent and null alike - but omitting is the honest
|
|
272
|
+
spelling of "not measured", and it keeps this method from having an opinion about what a
|
|
273
|
+
null means.
|
|
274
|
+
|
|
275
|
+
The field list is **derived** from this dataclass rather than written out. That is
|
|
276
|
+
requirement 6.7 from the writing side: a field added to the library joins this document
|
|
277
|
+
instead of being silently dropped, which is the mirror of the defect the control plane's
|
|
278
|
+
reader had - a hand-written list, against a dataclass where every field has a default, so
|
|
279
|
+
the next field would have been recorded as measured.
|
|
280
|
+
|
|
281
|
+
No float is formatted here and none is rounded. Every number in this document is either a
|
|
282
|
+
ratio of two integers or a bucket edge divided by a million, so two languages compute the
|
|
283
|
+
same IEEE 754 double - see :meth:`Window.as_record` for why that matters and where the
|
|
284
|
+
limit of the claim is.
|
|
285
|
+
"""
|
|
286
|
+
body: dict[str, Any] = {}
|
|
287
|
+
for name in MEASURED_FIELDS:
|
|
288
|
+
value = getattr(self, name)
|
|
289
|
+
if value is None:
|
|
290
|
+
continue
|
|
291
|
+
body[name] = dict(value) if isinstance(value, Mapping) else value
|
|
292
|
+
body["missing"] = sorted(self.missing)
|
|
293
|
+
body["complete"] = self.complete
|
|
294
|
+
return body
|
|
295
|
+
|
|
296
|
+
|
|
297
|
+
MEASURED_FIELDS: tuple[str, ...] = tuple(
|
|
298
|
+
spec.name
|
|
299
|
+
for spec in dataclass_fields(GroupFeatures)
|
|
300
|
+
if spec.name not in ("missing", "complete")
|
|
301
|
+
)
|
|
302
|
+
"""Every field of :class:`GroupFeatures` that carries a measurement.
|
|
303
|
+
|
|
304
|
+
Derived from the dataclass, and the two names excluded are bookkeeping *about* the measurement
|
|
305
|
+
rather than part of it. Both this module and the control plane's reader work from this same source,
|
|
306
|
+
so a field added to the feature vector cannot be written by one side and ignored by the other.
|
|
307
|
+
"""
|
|
308
|
+
|
|
309
|
+
|
|
310
|
+
@dataclass(frozen=True)
|
|
311
|
+
class Window:
|
|
312
|
+
"""One aggregation period, ready to send.
|
|
313
|
+
|
|
314
|
+
``complete`` is false when the library could not reach the control plane for part of the period,
|
|
315
|
+
or when the buffer dropped windows. A window that is not complete is still sent - the planner
|
|
316
|
+
needs to know traffic existed - but it may not be used to justify a migration.
|
|
317
|
+
"""
|
|
318
|
+
|
|
319
|
+
model_version: str
|
|
320
|
+
started_ns: int
|
|
321
|
+
ended_ns: int
|
|
322
|
+
shapes: Sequence[ShapeStats]
|
|
323
|
+
complete: bool = True
|
|
324
|
+
dropped_windows: int = 0
|
|
325
|
+
fanned: Sequence[FanOutStats] = ()
|
|
326
|
+
"""What the fan-out to each derived copy did. Empty when the group has no copy, which is the
|
|
327
|
+
ordinary case - a copy exists during a migration and while a derived materialisation is in the
|
|
328
|
+
map, not otherwise."""
|
|
329
|
+
|
|
330
|
+
def copies(self, group: str) -> tuple[CopyFreshness, ...]:
|
|
331
|
+
"""How far behind each of this group's derived copies ran, sorted by materialisation.
|
|
332
|
+
|
|
333
|
+
Read out of the histogram rather than stored, like every other percentile here. A stored
|
|
334
|
+
percentile is a second copy of a fact that changes when the samples do.
|
|
335
|
+
"""
|
|
336
|
+
return tuple(
|
|
337
|
+
CopyFreshness(
|
|
338
|
+
group=stats.group,
|
|
339
|
+
materialization=stats.materialization,
|
|
340
|
+
writes=stats.writes,
|
|
341
|
+
failures=stats.failures,
|
|
342
|
+
lag_p50_ms=stats.latency.percentile_ms(0.50),
|
|
343
|
+
lag_p99_ms=stats.latency.percentile_ms(0.99),
|
|
344
|
+
)
|
|
345
|
+
for stats in sorted(
|
|
346
|
+
(s for s in self.fanned if s.group == group),
|
|
347
|
+
key=lambda s: s.materialization,
|
|
348
|
+
)
|
|
349
|
+
)
|
|
350
|
+
|
|
351
|
+
def features(self, group: str, *, has_time_dimension: bool = False) -> GroupFeatures:
|
|
352
|
+
"""Fold this window's records for one group into the planner's feature vector."""
|
|
353
|
+
records = [s for s in self.shapes if s.group == group]
|
|
354
|
+
if not records:
|
|
355
|
+
# `no_traffic` is the *reason*, and the unknown fields are named too - by the same
|
|
356
|
+
# derivation as the branch below, because two branches of one function computing
|
|
357
|
+
# `missing` by two different rules is the defect this set exists to prevent. An empty
|
|
358
|
+
# window really did not measure a read/write ratio, and saying only "no traffic" leaves
|
|
359
|
+
# a reader to work out which fields that implies.
|
|
360
|
+
return _with_missing(
|
|
361
|
+
GroupFeatures(complete=self.complete, distinct_shapes=0), also={"no_traffic"}
|
|
362
|
+
)
|
|
363
|
+
|
|
364
|
+
writes = sum(s.calls for s in records if s.kind in WRITE_KINDS)
|
|
365
|
+
reads = sum(s.calls for s in records if s.kind not in WRITE_KINDS)
|
|
366
|
+
calls = writes + reads
|
|
367
|
+
|
|
368
|
+
latency = Histogram()
|
|
369
|
+
for record in records:
|
|
370
|
+
latency.merge(record.latency)
|
|
371
|
+
|
|
372
|
+
mix: dict[str, float] = {}
|
|
373
|
+
for record in records:
|
|
374
|
+
mix[record.kind] = mix.get(record.kind, 0.0) + record.calls
|
|
375
|
+
mix = {kind: hits / calls for kind, hits in sorted(mix.items())} if calls else {}
|
|
376
|
+
|
|
377
|
+
# A failed call counts in the latency histogram and **not** here, and the two answers have
|
|
378
|
+
# different reasons rather than one convention. A failure took time, so dropping it from
|
|
379
|
+
# the histogram would flatter a window precisely when the engine is in trouble. It returned
|
|
380
|
+
# no rows because it failed rather than because the data is sparse, so averaging that zero
|
|
381
|
+
# in understates how many rows a read of this shape returns - and `error_share` already
|
|
382
|
+
# carries the failure rate, so smearing it into a second feature is one fact in two places.
|
|
383
|
+
# A shape whose every call failed contributes nothing rather than a zero. `telemetry/008`.
|
|
384
|
+
read_records = [
|
|
385
|
+
s for s in records if s.kind not in WRITE_KINDS and s.calls > s.errors
|
|
386
|
+
]
|
|
387
|
+
cardinalities = sorted(s.rows / (s.calls - s.errors) for s in read_records)
|
|
388
|
+
|
|
389
|
+
pk_calls = sum(s.calls for s in records if s.kind == "point_read")
|
|
390
|
+
errors = sum(s.errors for s in records)
|
|
391
|
+
|
|
392
|
+
measured = GroupFeatures(
|
|
393
|
+
calls=calls,
|
|
394
|
+
read_write_ratio=(reads / writes) if writes else None,
|
|
395
|
+
shape_mix=mix,
|
|
396
|
+
latency_p50_ms=latency.percentile_ms(0.5),
|
|
397
|
+
latency_p99_ms=latency.percentile_ms(0.99),
|
|
398
|
+
result_cardinality_p50=_at(cardinalities, 0.5),
|
|
399
|
+
result_cardinality_p99=_at(cardinalities, 0.99),
|
|
400
|
+
pk_access_share=(pk_calls / calls) if calls else None,
|
|
401
|
+
has_time_dimension=has_time_dimension,
|
|
402
|
+
distinct_shapes=len(records),
|
|
403
|
+
error_share=(errors / calls) if calls else None,
|
|
404
|
+
complete=self.complete,
|
|
405
|
+
)
|
|
406
|
+
return _with_missing(measured)
|
|
407
|
+
|
|
408
|
+
|
|
409
|
+
|
|
410
|
+
|
|
411
|
+
def as_record(self, model: LogicalModel) -> dict[str, Any]:
|
|
412
|
+
"""This window as the document the control plane reads. Numbers, never rows.
|
|
413
|
+
|
|
414
|
+
**The artefact this library existed without.** Everything above measured traffic and
|
|
415
|
+
nothing turned a window into the file we are handed - so the walkthrough's step 8, "the
|
|
416
|
+
library measures traffic and hands us a window", was a literal dictionary typed into the
|
|
417
|
+
example, and every client in every language would have written their own. Two clients of
|
|
418
|
+
one model would then produce two documents from identical traffic, and the planner would
|
|
419
|
+
score whichever one it was given. Same shape of gap as the two before it: both halves
|
|
420
|
+
worked and nothing joined them.
|
|
421
|
+
|
|
422
|
+
The model is required and it is checked. Only one fact is read from it - whether a group
|
|
423
|
+
carries a time dimension, which is decided by declared *type* and never by a field's name -
|
|
424
|
+
but a window serialised against the wrong model would attach that fact to the wrong groups
|
|
425
|
+
and claim `has_time_dimension: false` for a group that has one. False is a claim; a
|
|
426
|
+
measurement this library cannot make has to be absent, which is what everything else in
|
|
427
|
+
here is careful about.
|
|
428
|
+
|
|
429
|
+
**This document is deliberately not canonical, and that needs saying because §1 of the
|
|
430
|
+
format contract rejects floating point outright.** Its reason is that a float's textual
|
|
431
|
+
form differs between languages, and almost every number here is a float. This is not
|
|
432
|
+
signed, not hashed and never compared for equality, so the rule it breaks does not apply -
|
|
433
|
+
but the thing that makes the family checkable at all is narrower and worth stating: every
|
|
434
|
+
number in this document is either **a ratio of two integers** or **a bucket edge divided by
|
|
435
|
+
a million**, and IEEE 754 requires division to be correctly rounded. So two languages
|
|
436
|
+
compute the same double from the same traffic even where they would print it differently,
|
|
437
|
+
and the `telemetry/` vectors compare numbers rather than bytes for exactly that reason.
|
|
438
|
+
"""
|
|
439
|
+
if model.version != self.model_version:
|
|
440
|
+
raise ValueError(
|
|
441
|
+
f"this window measured model version {self.model_version} and it is being "
|
|
442
|
+
f"serialised against {model.version}. Only one fact is read from the model here - "
|
|
443
|
+
f"whether a group has a time dimension - and reading it from another model would "
|
|
444
|
+
f"attach it to the wrong groups. Keep the model the recorder was created for, or "
|
|
445
|
+
f"drop the window: a window measured against a model that no longer exists cannot "
|
|
446
|
+
f"be scored against the one that does."
|
|
447
|
+
)
|
|
448
|
+
by_name = {group.name: group for group in colocation_groups(model)}
|
|
449
|
+
named = sorted({stats.group for stats in self.shapes} | {s.group for s in self.fanned})
|
|
450
|
+
unknown = [name for name in named if name not in by_name]
|
|
451
|
+
if unknown:
|
|
452
|
+
raise ValueError(
|
|
453
|
+
f"this window has records for groups this model does not have: {unknown}. The "
|
|
454
|
+
f"recorder is given a model version and the groups come from the operations it "
|
|
455
|
+
f"observed, so this is a session routing a model other than the one measured."
|
|
456
|
+
)
|
|
457
|
+
|
|
458
|
+
groups: dict[str, Any] = {}
|
|
459
|
+
for name in named:
|
|
460
|
+
record = self.features(
|
|
461
|
+
name, has_time_dimension=has_time_dimension(model, by_name[name])
|
|
462
|
+
).as_record()
|
|
463
|
+
copies = [copy.as_record() for copy in self.copies(name)]
|
|
464
|
+
if copies:
|
|
465
|
+
# Absent rather than empty when the group has no derived copy, for the reason
|
|
466
|
+
# `also_write` is absent rather than empty in a map: absent says "not doing this"
|
|
467
|
+
# and an empty list says "considered and found none", which is a stronger claim.
|
|
468
|
+
record["copies"] = copies
|
|
469
|
+
groups[name] = record
|
|
470
|
+
|
|
471
|
+
return {
|
|
472
|
+
"model_version": self.model_version,
|
|
473
|
+
"complete": self.complete,
|
|
474
|
+
"dropped_windows": self.dropped_windows,
|
|
475
|
+
"groups": groups,
|
|
476
|
+
}
|
|
477
|
+
|
|
478
|
+
|
|
479
|
+
def _with_missing(
|
|
480
|
+
measured: GroupFeatures, *, also: Collection[str] = ()
|
|
481
|
+
) -> GroupFeatures:
|
|
482
|
+
"""The same features with ``missing`` **derived** from which values came out unknown.
|
|
483
|
+
|
|
484
|
+
Derived rather than listed alongside them, so the two cannot disagree - and they did.
|
|
485
|
+
``time_filtered_share`` is a field this library never measures and it was left null and absent
|
|
486
|
+
from this set, which breaks the one promise the set makes: that a reader never has to infer
|
|
487
|
+
absence from a null. Four other fields were named by hand and would have stayed the only ones.
|
|
488
|
+
|
|
489
|
+
What is unknown and why, because a derivation hides the reasons. Engine-side sizes
|
|
490
|
+
(``total_bytes``, ``index_to_table_ratio``) need a catalogue read, which is an adapter
|
|
491
|
+
capability rather than a measurement. ``daily_growth_bytes`` and ``write_burstiness`` need two
|
|
492
|
+
samples over time and a window is one. ``time_filtered_share`` would need the *arguments* of a
|
|
493
|
+
call, and this library records shapes and never values. Zero bytes and unknown bytes lead a
|
|
494
|
+
planner to opposite conclusions, which is the whole reason for naming them rather than
|
|
495
|
+
defaulting them.
|
|
496
|
+
|
|
497
|
+
``also`` carries reasons that are not field names - ``no_traffic`` is the only one - because a
|
|
498
|
+
reader of the window needs to know the difference between "this group was idle" and "this
|
|
499
|
+
field is not measurable".
|
|
500
|
+
"""
|
|
501
|
+
return replace(
|
|
502
|
+
measured,
|
|
503
|
+
missing=frozenset(also)
|
|
504
|
+
| frozenset(name for name in MEASURED_FIELDS if getattr(measured, name) is None),
|
|
505
|
+
)
|
|
506
|
+
|
|
507
|
+
|
|
508
|
+
def _at(ordered: list[float], fraction: float) -> float | None:
|
|
509
|
+
"""Nearest-rank: the smallest sample at least ``fraction`` of the data is not above.
|
|
510
|
+
|
|
511
|
+
**The same rank rule as** :meth:`Histogram.percentile_ms`, and it did not used to be. This
|
|
512
|
+
function selected ``ordered[int(len * fraction)]`` - the floor - while the histogram takes the
|
|
513
|
+
first bucket whose cumulative count reaches ``fraction * count``, which is the ceiling. The two
|
|
514
|
+
agree on every sample set with an odd count, and every case in ``telemetry/`` had one, so one
|
|
515
|
+
window document carried two percentile conventions and nothing could see it: a third
|
|
516
|
+
implementation swapped one for the other and the whole family stayed green.
|
|
517
|
+
|
|
518
|
+
They differ exactly when ``fraction * len`` is an integer. Two samples at p50 is that case:
|
|
519
|
+
``[3.0, 300.0]`` reported 300 here and the *lower* bucket in the histogram, for the same
|
|
520
|
+
reason applied twice in opposite directions. ``telemetry/007`` pins it.
|
|
521
|
+
"""
|
|
522
|
+
if not ordered:
|
|
523
|
+
return None
|
|
524
|
+
rank = max(1, math.ceil(fraction * len(ordered))) - 1
|
|
525
|
+
return ordered[min(len(ordered) - 1, rank)]
|
|
526
|
+
|
|
527
|
+
|
|
528
|
+
# Types that make a field a time dimension. Recognised by **type**, never by name - see below.
|
|
529
|
+
_TIME_TYPES = frozenset({"date", "timestamp", "timestamptz"})
|
|
530
|
+
|
|
531
|
+
|
|
532
|
+
def has_time_dimension(model: LogicalModel, group: Group) -> bool:
|
|
533
|
+
"""Does any entity in this group carry a time dimension?
|
|
534
|
+
|
|
535
|
+
Decided by the declared type and never by the field's name. Two reasons, and the second is the
|
|
536
|
+
one that made it a rule rather than a preference.
|
|
537
|
+
|
|
538
|
+
A name is not evidence: `created_at` typed as a string is a string, and treating it as a
|
|
539
|
+
timestamp would have the planner recommend time partitioning on a column no engine can
|
|
540
|
+
range-scan usefully.
|
|
541
|
+
|
|
542
|
+
And a client may hash identifier names (requirement 11.3) so that we never see them. A
|
|
543
|
+
derivation that reads names would silently produce different answers with hashing on and off -
|
|
544
|
+
which is exactly the property the conformance vector for hashed models exists to forbid.
|
|
545
|
+
Deciding by type also means models written in other natural languages work identically, which is
|
|
546
|
+
a free consequence of getting this right for the other reason.
|
|
547
|
+
"""
|
|
548
|
+
for member in group.members:
|
|
549
|
+
spec = model.entity(member)
|
|
550
|
+
for spec_field in spec.fields:
|
|
551
|
+
if spec_field.type in _TIME_TYPES:
|
|
552
|
+
return True
|
|
553
|
+
return False
|
|
554
|
+
|
|
555
|
+
|
|
556
|
+
class Recorder:
|
|
557
|
+
"""Accumulates records, rolls windows, and drops telemetry rather than anything else.
|
|
558
|
+
|
|
559
|
+
Thread safety is a single lock taken only when a shape is first seen or a window rolls - not on
|
|
560
|
+
every record. Counters for an existing shape are updated without it: two threads racing on the
|
|
561
|
+
same shape can lose a call from the count, and that is the right trade. Telemetry informs a
|
|
562
|
+
placement decision made over days; a lock on the hot path would cost every operation forever.
|
|
563
|
+
"""
|
|
564
|
+
|
|
565
|
+
def __init__(self, model_version: str, *, max_windows: int = 64) -> None:
|
|
566
|
+
self._model_version = model_version
|
|
567
|
+
self._lock = threading.Lock()
|
|
568
|
+
self._current: dict[str, ShapeStats] = {}
|
|
569
|
+
self._fanned: dict[tuple[str, str], FanOutStats] = {}
|
|
570
|
+
self._started_ns = _now()
|
|
571
|
+
self._windows: deque[Window] = deque(maxlen=max_windows)
|
|
572
|
+
self._dropped = 0
|
|
573
|
+
self._incomplete = False
|
|
574
|
+
|
|
575
|
+
# --- recording ---------------------------------------------------------------------------
|
|
576
|
+
|
|
577
|
+
def record(
|
|
578
|
+
self,
|
|
579
|
+
*,
|
|
580
|
+
shape_id: str,
|
|
581
|
+
group: str,
|
|
582
|
+
entity: str,
|
|
583
|
+
kind: str,
|
|
584
|
+
nanoseconds: int,
|
|
585
|
+
rows: int = 0,
|
|
586
|
+
failed: bool = False,
|
|
587
|
+
) -> None:
|
|
588
|
+
"""Record one operation. Never raises, never blocks on the common path."""
|
|
589
|
+
guard(
|
|
590
|
+
"telemetry.record",
|
|
591
|
+
lambda: self._record(shape_id, group, entity, kind, nanoseconds, rows, failed),
|
|
592
|
+
)
|
|
593
|
+
|
|
594
|
+
def _record(
|
|
595
|
+
self,
|
|
596
|
+
shape_id: str,
|
|
597
|
+
group: str,
|
|
598
|
+
entity: str,
|
|
599
|
+
kind: str,
|
|
600
|
+
nanoseconds: int,
|
|
601
|
+
rows: int,
|
|
602
|
+
failed: bool,
|
|
603
|
+
) -> None:
|
|
604
|
+
stats = self._current.get(shape_id)
|
|
605
|
+
if stats is None:
|
|
606
|
+
with self._lock:
|
|
607
|
+
stats = self._current.get(shape_id)
|
|
608
|
+
if stats is None:
|
|
609
|
+
stats = ShapeStats(
|
|
610
|
+
shape_id=shape_id, group=group, entity=entity, kind=kind,
|
|
611
|
+
call_site=_call_site(),
|
|
612
|
+
)
|
|
613
|
+
self._current[shape_id] = stats
|
|
614
|
+
stats.record(nanoseconds, rows, failed)
|
|
615
|
+
|
|
616
|
+
def record_fan_out(
|
|
617
|
+
self, *, group: str, materialization: str, nanoseconds: int, failed: bool = False
|
|
618
|
+
) -> None:
|
|
619
|
+
"""Record one write to one derived copy. Never raises, never blocks on the common path.
|
|
620
|
+
|
|
621
|
+
A separate entry point from :meth:`record` rather than a shape kind, because a fan-out is
|
|
622
|
+
not an operation the application asked for - see :class:`FanOutStats`.
|
|
623
|
+
"""
|
|
624
|
+
guard(
|
|
625
|
+
"telemetry.record_fan_out",
|
|
626
|
+
lambda: self._record_fan_out(group, materialization, nanoseconds, failed),
|
|
627
|
+
)
|
|
628
|
+
|
|
629
|
+
def _record_fan_out(
|
|
630
|
+
self, group: str, materialization: str, nanoseconds: int, failed: bool
|
|
631
|
+
) -> None:
|
|
632
|
+
key = (group, materialization)
|
|
633
|
+
stats = self._fanned.get(key)
|
|
634
|
+
if stats is None:
|
|
635
|
+
with self._lock:
|
|
636
|
+
stats = self._fanned.get(key)
|
|
637
|
+
if stats is None:
|
|
638
|
+
stats = FanOutStats(group=group, materialization=materialization)
|
|
639
|
+
self._fanned[key] = stats
|
|
640
|
+
stats.record(nanoseconds, failed)
|
|
641
|
+
|
|
642
|
+
# --- windows -----------------------------------------------------------------------------
|
|
643
|
+
|
|
644
|
+
def roll(self) -> Window | None:
|
|
645
|
+
"""Close the current period and queue it.
|
|
646
|
+
|
|
647
|
+
Returns the window, or None if nothing was recorded in it.
|
|
648
|
+
"""
|
|
649
|
+
return guard("telemetry.roll", self._roll)
|
|
650
|
+
|
|
651
|
+
def _roll(self) -> Window | None:
|
|
652
|
+
with self._lock:
|
|
653
|
+
if not self._current and not self._fanned:
|
|
654
|
+
# `not self._current` alone would drop a window holding only fan-out records. That
|
|
655
|
+
# cannot arise from an application write - a write records a shape and then fans
|
|
656
|
+
# out - but it can from a backfill replaying rows, and a window silently discarded
|
|
657
|
+
# is the shape of gap that makes a client's copy look healthier than it is.
|
|
658
|
+
return None
|
|
659
|
+
window = Window(
|
|
660
|
+
model_version=self._model_version,
|
|
661
|
+
started_ns=self._started_ns,
|
|
662
|
+
ended_ns=_now(),
|
|
663
|
+
shapes=tuple(self._current.values()),
|
|
664
|
+
complete=not self._incomplete,
|
|
665
|
+
dropped_windows=self._dropped,
|
|
666
|
+
fanned=tuple(self._fanned.values()),
|
|
667
|
+
)
|
|
668
|
+
self._current = {}
|
|
669
|
+
self._fanned = {}
|
|
670
|
+
self._started_ns = _now()
|
|
671
|
+
self._incomplete = False
|
|
672
|
+
|
|
673
|
+
# A full buffer drops the oldest window and says so in the next one. Telemetry is the
|
|
674
|
+
# thing that gets lost when we run out of room - never a write, never an operation.
|
|
675
|
+
if len(self._windows) == self._windows.maxlen:
|
|
676
|
+
self._dropped += 1
|
|
677
|
+
log("sde.telemetry.dropped", dropped=self._dropped)
|
|
678
|
+
self._windows.append(window)
|
|
679
|
+
log(
|
|
680
|
+
"sde.telemetry.window_closed",
|
|
681
|
+
model_version=window.model_version,
|
|
682
|
+
shapes=len(window.shapes),
|
|
683
|
+
complete=window.complete,
|
|
684
|
+
)
|
|
685
|
+
return window
|
|
686
|
+
|
|
687
|
+
def pending(self) -> tuple[Window, ...]:
|
|
688
|
+
with self._lock:
|
|
689
|
+
return tuple(self._windows)
|
|
690
|
+
|
|
691
|
+
def mark_incomplete(self) -> None:
|
|
692
|
+
"""Called by the application when part of the current period was not recorded.
|
|
693
|
+
|
|
694
|
+
The flag travels to us in :attr:`Window.complete` and the planner reads it, because a
|
|
695
|
+
window missing a slice of the traffic must not be scored as though it were the whole
|
|
696
|
+
period: an application that was down for six of the last twenty-four hours looks
|
|
697
|
+
idle-and-cheap on figures nobody marked as partial.
|
|
698
|
+
|
|
699
|
+
Nothing here decides when that happened. The application collects these windows and hands
|
|
700
|
+
them to us; only it knows whether a collection was skipped.
|
|
701
|
+
"""
|
|
702
|
+
self._incomplete = True
|
|
703
|
+
|
|
704
|
+
def acknowledge(self, count: int) -> None:
|
|
705
|
+
"""Drop the oldest ``count`` windows once the application has taken them.
|
|
706
|
+
|
|
707
|
+
The delivering party is the client's own process, and there is no other. This library
|
|
708
|
+
never opens a connection to us - the buffer is read with :meth:`pending`, written wherever
|
|
709
|
+
the application writes it, and that file is what we are handed.
|
|
710
|
+
"""
|
|
711
|
+
with self._lock:
|
|
712
|
+
for _ in range(min(count, len(self._windows))):
|
|
713
|
+
self._windows.popleft()
|
|
714
|
+
|
|
715
|
+
|
|
716
|
+
def _now() -> int:
|
|
717
|
+
return int(__import__("time").perf_counter_ns())
|
|
718
|
+
|
|
719
|
+
|
|
720
|
+
def _call_site() -> str | None:
|
|
721
|
+
"""Where in the client's code this shape is used. Walked once per shape, never per call.
|
|
722
|
+
|
|
723
|
+
Best effort by design: a frame walk is not free and a missing call site costs a little
|
|
724
|
+
diagnostic value, while a frame walk on every operation would cost the latency budget.
|
|
725
|
+
"""
|
|
726
|
+
|
|
727
|
+
def walk() -> str | None:
|
|
728
|
+
frame: Any = sys._getframe(1)
|
|
729
|
+
while frame is not None:
|
|
730
|
+
name = frame.f_globals.get("__name__", "")
|
|
731
|
+
if not name.startswith("sde"):
|
|
732
|
+
return f"{frame.f_code.co_filename}:{frame.f_lineno}"
|
|
733
|
+
frame = frame.f_back
|
|
734
|
+
return None
|
|
735
|
+
|
|
736
|
+
return guard("telemetry.call_site", walk)
|