@smart-data-engines/sde 0.1.0-dev.0

This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
Files changed (131) hide show
  1. package/LICENSE +201 -0
  2. package/NOTICE +13 -0
  3. package/README.md +153 -0
  4. package/bin/weather.mjs +32 -0
  5. package/dist/_usage.d.ts +30 -0
  6. package/dist/_usage.js +194 -0
  7. package/dist/_usage.js.map +1 -0
  8. package/dist/bulk.d.ts +9 -0
  9. package/dist/bulk.js +83 -0
  10. package/dist/bulk.js.map +1 -0
  11. package/dist/canonical.d.ts +46 -0
  12. package/dist/canonical.js +150 -0
  13. package/dist/canonical.js.map +1 -0
  14. package/dist/capabilities.d.ts +48 -0
  15. package/dist/capabilities.js +62 -0
  16. package/dist/capabilities.js.map +1 -0
  17. package/dist/cutover.d.ts +36 -0
  18. package/dist/cutover.js +219 -0
  19. package/dist/cutover.js.map +1 -0
  20. package/dist/demo/model.d.ts +28 -0
  21. package/dist/demo/model.js +40 -0
  22. package/dist/demo/model.js.map +1 -0
  23. package/dist/demo/project.d.ts +19 -0
  24. package/dist/demo/project.js +128 -0
  25. package/dist/demo/project.js.map +1 -0
  26. package/dist/demo/weather.d.ts +73 -0
  27. package/dist/demo/weather.js +334 -0
  28. package/dist/demo/weather.js.map +1 -0
  29. package/dist/engines/_clickhouse-connection.d.ts +17 -0
  30. package/dist/engines/_clickhouse-connection.js +182 -0
  31. package/dist/engines/_clickhouse-connection.js.map +1 -0
  32. package/dist/engines/_tls-peer-identity.d.ts +2 -0
  33. package/dist/engines/_tls-peer-identity.js +23 -0
  34. package/dist/engines/_tls-peer-identity.js.map +1 -0
  35. package/dist/engines/_write-fences.d.ts +51 -0
  36. package/dist/engines/_write-fences.js +189 -0
  37. package/dist/engines/_write-fences.js.map +1 -0
  38. package/dist/engines/clickhouse.d.ts +193 -0
  39. package/dist/engines/clickhouse.js +899 -0
  40. package/dist/engines/clickhouse.js.map +1 -0
  41. package/dist/engines/postgres.d.ts +293 -0
  42. package/dist/engines/postgres.js +981 -0
  43. package/dist/engines/postgres.js.map +1 -0
  44. package/dist/errors.d.ts +89 -0
  45. package/dist/errors.js +90 -0
  46. package/dist/errors.js.map +1 -0
  47. package/dist/frozen-verification.d.ts +26 -0
  48. package/dist/frozen-verification.js +67 -0
  49. package/dist/frozen-verification.js.map +1 -0
  50. package/dist/generation.d.ts +34 -0
  51. package/dist/generation.js +81 -0
  52. package/dist/generation.js.map +1 -0
  53. package/dist/groups.d.ts +17 -0
  54. package/dist/groups.js +66 -0
  55. package/dist/groups.js.map +1 -0
  56. package/dist/hashing.d.ts +68 -0
  57. package/dist/hashing.js +146 -0
  58. package/dist/hashing.js.map +1 -0
  59. package/dist/in-place-index.d.ts +43 -0
  60. package/dist/in-place-index.js +272 -0
  61. package/dist/in-place-index.js.map +1 -0
  62. package/dist/index.d.ts +79 -0
  63. package/dist/index.js +64 -0
  64. package/dist/index.js.map +1 -0
  65. package/dist/inspection.d.ts +19 -0
  66. package/dist/inspection.js +31 -0
  67. package/dist/inspection.js.map +1 -0
  68. package/dist/internal.d.ts +42 -0
  69. package/dist/internal.js +56 -0
  70. package/dist/internal.js.map +1 -0
  71. package/dist/layout.d.ts +36 -0
  72. package/dist/layout.js +62 -0
  73. package/dist/layout.js.map +1 -0
  74. package/dist/migration.d.ts +197 -0
  75. package/dist/migration.js +592 -0
  76. package/dist/migration.js.map +1 -0
  77. package/dist/model.d.ts +93 -0
  78. package/dist/model.js +313 -0
  79. package/dist/model.js.map +1 -0
  80. package/dist/physical.d.ts +128 -0
  81. package/dist/physical.js +421 -0
  82. package/dist/physical.js.map +1 -0
  83. package/dist/placement.d.ts +157 -0
  84. package/dist/placement.js +651 -0
  85. package/dist/placement.js.map +1 -0
  86. package/dist/provisioning.d.ts +6 -0
  87. package/dist/provisioning.js +45 -0
  88. package/dist/provisioning.js.map +1 -0
  89. package/dist/query.d.ts +68 -0
  90. package/dist/query.js +340 -0
  91. package/dist/query.js.map +1 -0
  92. package/dist/routing.d.ts +25 -0
  93. package/dist/routing.js +35 -0
  94. package/dist/routing.js.map +1 -0
  95. package/dist/schema.d.ts +110 -0
  96. package/dist/schema.js +337 -0
  97. package/dist/schema.js.map +1 -0
  98. package/dist/session.d.ts +195 -0
  99. package/dist/session.js +870 -0
  100. package/dist/session.js.map +1 -0
  101. package/dist/shapes.d.ts +30 -0
  102. package/dist/shapes.js +112 -0
  103. package/dist/shapes.js.map +1 -0
  104. package/dist/staging.d.ts +29 -0
  105. package/dist/staging.js +214 -0
  106. package/dist/staging.js.map +1 -0
  107. package/dist/telemetry.d.ts +468 -0
  108. package/dist/telemetry.js +872 -0
  109. package/dist/telemetry.js.map +1 -0
  110. package/dist/testing/loader.d.ts +38 -0
  111. package/dist/testing/loader.js +86 -0
  112. package/dist/testing/loader.js.map +1 -0
  113. package/dist/testing/memory.d.ts +131 -0
  114. package/dist/testing/memory.js +311 -0
  115. package/dist/testing/memory.js.map +1 -0
  116. package/dist/timestamp.d.ts +20 -0
  117. package/dist/timestamp.js +89 -0
  118. package/dist/timestamp.js.map +1 -0
  119. package/dist/types.d.ts +79 -0
  120. package/dist/types.js +100 -0
  121. package/dist/types.js.map +1 -0
  122. package/dist/verification.d.ts +41 -0
  123. package/dist/verification.js +169 -0
  124. package/dist/verification.js.map +1 -0
  125. package/dist/watermark.d.ts +103 -0
  126. package/dist/watermark.js +170 -0
  127. package/dist/watermark.js.map +1 -0
  128. package/dist/write-fence.d.ts +58 -0
  129. package/dist/write-fence.js +225 -0
  130. package/dist/write-fence.js.map +1 -0
  131. package/package.json +86 -0
@@ -0,0 +1,872 @@
1
+ /**
2
+ * Measuring what the application actually does, without ever seeing what it does it to.
3
+ *
4
+ * This is the input a placement decision is made from, so its shape matters more than its
5
+ * precision. Three constraints shaped everything here.
6
+ *
7
+ * **It carries no values.** A record is keyed by an operation shape, which is assembled from the
8
+ * structure of a call and never sees its arguments. There is no code path by which a customer's row
9
+ * reaches a telemetry record, which is why this file can be read by a client and believed.
10
+ *
11
+ * **It cannot cost anything.** Routing already has a one percent budget for the whole library and
12
+ * recording happens on the same path. So: no string formatting per record, no stack walking, and a
13
+ * histogram rather than a list of samples. There is no lock, and that is not a shortcut - this
14
+ * runtime has one thread per record, so the trade the reference implementation makes (a lock taken
15
+ * only when a shape is first seen) has nothing to buy here.
16
+ *
17
+ * **It cannot fail the caller.** Every entry point goes through `guard`. A bug in an aggregation
18
+ * counter must not take down somebody's request - and the failure is counted, so it is not
19
+ * invisible either.
20
+ *
21
+ * The histogram deserves a word, because it is the one deliberate loss of precision. Latency lands
22
+ * in exponential buckets, so a percentile read out of it is approximate - within one bucket width,
23
+ * which is a factor of two at the extremes. That is ample for the decision it feeds: a planner
24
+ * cares whether a group's reads are microseconds or milliseconds, not whether p99 is 412 or 431
25
+ * microseconds.
26
+ */
27
+ import { compareCodePoints } from './canonical.js';
28
+ import { colocationGroups } from './groups.js';
29
+ import { guard } from './internal.js';
30
+ import { WRITE_KINDS, enumerateShapes, shapeId } from './shapes.js';
31
+ /** 1 microsecond to about 17 s, doubling. Enough to tell a cache hit from a full scan. */
32
+ export const BUCKET_COUNT = 25;
33
+ export const BUCKET_BASE_NS = 1000;
34
+ /** Exponential-bucket histogram. Fixed memory, O(1) record, approximate percentiles. */
35
+ export class Histogram {
36
+ buckets = new Array(BUCKET_COUNT).fill(0);
37
+ count = 0;
38
+ total = 0;
39
+ /**
40
+ * Put one duration in its bucket. **Integer arithmetic only, and that is the point.**
41
+ *
42
+ * The obvious form is `Math.floor(Math.log2(ns / BUCKET_BASE_NS)) + 1`, which is the same
43
+ * function and the wrong way to compute it in a library that has to agree with another
44
+ * implementation. `log2` is not required by IEEE 754 to be correctly rounded, so two runtimes may
45
+ * differ in the last bit - and one bit at a power-of-two boundary is a different bucket, which is
46
+ * a different p99 for identical traffic, in a number a placement decision is made from.
47
+ *
48
+ * The bit length of the integer quotient is exact everywhere. `Math.clz32` is defined on the
49
+ * 32-bit value, and the quotient is below 2^24 in every case that is not clamped, so the clamp is
50
+ * checked first rather than relying on the coercion.
51
+ *
52
+ * **The vectors cannot see the difference, and that is worth saying rather than implying.**
53
+ * Measured: the logarithm form passes every vector in `telemetry/`, because glibc's `log2` and
54
+ * V8's are both exact at a power of two - the two runtimes we have agree, and the hazard is a
55
+ * *third* libm that does not. A property no output can distinguish on the machines available is
56
+ * not one a vector can hold, so it is held statically: `telemetry.test.ts` refuses a logarithm in
57
+ * this file.
58
+ */
59
+ record(nanoseconds) {
60
+ this.count += 1;
61
+ this.total += nanoseconds;
62
+ if (nanoseconds < BUCKET_BASE_NS) {
63
+ this.buckets[0] = (this.buckets[0] ?? 0) + 1;
64
+ return;
65
+ }
66
+ const scaled = Math.floor(nanoseconds / BUCKET_BASE_NS);
67
+ const bits = scaled >= 2 ** (BUCKET_COUNT - 1) ? BUCKET_COUNT : 32 - Math.clz32(scaled);
68
+ const index = Math.min(BUCKET_COUNT - 1, bits);
69
+ // `?? 0` rather than `+= 1`: `noUncheckedIndexedAccess` types an indexed read as possibly
70
+ // undefined, and the array is allocated full - so this is the compiler's price for a setting
71
+ // that has caught real bugs elsewhere in this library, paid on a hot path, deliberately.
72
+ this.buckets[index] = (this.buckets[index] ?? 0) + 1;
73
+ }
74
+ /**
75
+ * Approximate percentile in milliseconds, or null if nothing was recorded.
76
+ *
77
+ * Returns the *upper* edge of the bucket the percentile falls in. Rounding up rather than
78
+ * interpolating is deliberate: a placement decision made on an optimistic latency figure is the
79
+ * wrong kind of wrong.
80
+ */
81
+ percentileMs(fraction) {
82
+ if (this.count === 0)
83
+ return null;
84
+ const target = fraction * this.count;
85
+ let seen = 0;
86
+ for (let index = 0; index < BUCKET_COUNT; index += 1) {
87
+ seen += this.buckets[index] ?? 0;
88
+ if (seen >= target)
89
+ return (BUCKET_BASE_NS * 2 ** index) / 1_000_000;
90
+ }
91
+ return null;
92
+ }
93
+ merge(other) {
94
+ for (let index = 0; index < BUCKET_COUNT; index += 1) {
95
+ this.buckets[index] = (this.buckets[index] ?? 0) + (other.buckets[index] ?? 0);
96
+ }
97
+ this.count += other.count;
98
+ this.total += other.total;
99
+ }
100
+ }
101
+ /** What was observed for one operation shape. No values, by construction. */
102
+ export class ShapeStats {
103
+ shapeId;
104
+ group;
105
+ entity;
106
+ kind;
107
+ calls = 0;
108
+ rows = 0;
109
+ errors = 0;
110
+ latency = new Histogram();
111
+ /**
112
+ * Calls by what they filtered on - equality fields and the bounded field (`''` for none) - names
113
+ * only, never values. Two reads of one shape can want different key orders; only an operation
114
+ * that takes a `where` reports this, so writes and point reads leave it empty.
115
+ */
116
+ filtered = new Map();
117
+ constructor(shapeId, group, entity, kind) {
118
+ this.shapeId = shapeId;
119
+ this.group = group;
120
+ this.entity = entity;
121
+ this.kind = kind;
122
+ }
123
+ record(nanoseconds, rows, failed, predicates) {
124
+ this.calls += 1;
125
+ this.rows += rows;
126
+ if (failed)
127
+ this.errors += 1;
128
+ this.latency.record(nanoseconds);
129
+ if (predicates !== undefined) {
130
+ const key = JSON.stringify([predicates.equal, predicates.ranged]);
131
+ const seen = this.filtered.get(key);
132
+ if (seen === undefined)
133
+ this.filtered.set(key, { predicates, calls: 1 });
134
+ else
135
+ seen.calls += 1;
136
+ }
137
+ }
138
+ }
139
+ /** Code point order of two predicate combinations: equality fields element by element, then range. */
140
+ function comparePredicates(a, b) {
141
+ const length = Math.min(a.equal.length, b.equal.length);
142
+ for (let index = 0; index < length; index += 1) {
143
+ const order = compareCodePoints(a.equal[index], b.equal[index]);
144
+ if (order !== 0)
145
+ return order;
146
+ }
147
+ if (a.equal.length !== b.equal.length)
148
+ return a.equal.length - b.equal.length;
149
+ return compareCodePoints(a.ranged, b.ranged);
150
+ }
151
+ /**
152
+ * What was observed writing one row to one derived copy. No values, by construction.
153
+ *
154
+ * Deliberately **not** a `ShapeStats`, and that is the load-bearing decision. A fan-out is not an
155
+ * operation the application asked for - it is the library keeping a copy current - so recording it
156
+ * as a shape would add a write to the very counters a placement is scored on: `read_write_ratio`
157
+ * would move because a copy exists, and a group with one copy would look twice as write-heavy as
158
+ * the same group without one. It is also not part of `GroupFeatures`: a copy's freshness does not
159
+ * score a placement, it reports the health of one already made.
160
+ */
161
+ export class FanOutStats {
162
+ group;
163
+ materialization;
164
+ writes = 0;
165
+ /** Rows that did not reach the copy. Absence, not lateness - see `CopyFreshness`. */
166
+ failures = 0;
167
+ latency = new Histogram();
168
+ constructor(group, materialization) {
169
+ this.group = group;
170
+ this.materialization = materialization;
171
+ }
172
+ record(nanoseconds, failed) {
173
+ this.writes += 1;
174
+ if (failed)
175
+ this.failures += 1;
176
+ // Recorded either way. A failed fan-out took time too, and dropping it would make the measured
177
+ // window look better precisely when the copy is in trouble.
178
+ this.latency.record(nanoseconds);
179
+ }
180
+ }
181
+ export function copyFreshnessRecord(copy) {
182
+ return {
183
+ group: copy.group,
184
+ materialization: copy.materialization,
185
+ writes: copy.writes,
186
+ failures: copy.failures,
187
+ lag_p50_ms: copy.lagP50Ms,
188
+ lag_p99_ms: copy.lagP99Ms,
189
+ complete: copy.complete,
190
+ };
191
+ }
192
+ /**
193
+ * The document key for every measured field, and the property that carries it.
194
+ *
195
+ * Two spellings because two things are being named: the **document** is the format, shared with
196
+ * every other implementation and therefore `snake_case`; the **property** is this language's, and a
197
+ * library that reads like transliterated Python is a worse library in TypeScript and no better an
198
+ * SDE one.
199
+ *
200
+ * The reference implementation derives its equivalent from the dataclass at runtime, which is not
201
+ * available here - types are erased before the code runs. So the list is the source and the type is
202
+ * checked against it, which is the same guarantee in the other direction and is enforced by the
203
+ * compiler rather than by a test: see `FIELD_LIST_IS_TOTAL`.
204
+ */
205
+ export const MEASURED_FIELDS = [
206
+ ['calls', 'calls'],
207
+ ['read_write_ratio', 'readWriteRatio'],
208
+ ['shape_mix', 'shapeMix'],
209
+ ['latency_p50_ms', 'latencyP50Ms'],
210
+ ['latency_p99_ms', 'latencyP99Ms'],
211
+ ['result_cardinality_p50', 'resultCardinalityP50'],
212
+ ['result_cardinality_p99', 'resultCardinalityP99'],
213
+ ['total_bytes', 'totalBytes'],
214
+ ['daily_growth_bytes', 'dailyGrowthBytes'],
215
+ ['index_to_table_ratio', 'indexToTableRatio'],
216
+ ['pk_access_share', 'pkAccessShare'],
217
+ ['has_time_dimension', 'hasTimeDimension'],
218
+ ['time_filtered_share', 'timeFilteredShare'],
219
+ ['distinct_shapes', 'distinctShapes'],
220
+ ['write_burstiness', 'writeBurstiness'],
221
+ ['error_share', 'errorShare'],
222
+ ];
223
+ /**
224
+ * A compile-time ratchet in both directions.
225
+ *
226
+ * Add a field to `GroupFeatures` and forget the list, and the second element stops being
227
+ * assignable; put a name in the list that is not a field, and the first does. `tsc` fails either
228
+ * way, which is where this belongs: the control plane's reader had a hand-written field list
229
+ * against a dataclass where every field has a default, so the next field added to the library would
230
+ * have been recorded as *measured* rather than missing. That defect had to be found by mutation.
231
+ * This one cannot exist.
232
+ */
233
+ export const FIELD_LIST_IS_TOTAL = [true, true];
234
+ const EMPTY_FEATURES = {
235
+ calls: 0,
236
+ readWriteRatio: null,
237
+ shapeMix: {},
238
+ latencyP50Ms: null,
239
+ latencyP99Ms: null,
240
+ resultCardinalityP50: null,
241
+ resultCardinalityP99: null,
242
+ totalBytes: null,
243
+ dailyGrowthBytes: null,
244
+ indexToTableRatio: null,
245
+ pkAccessShare: null,
246
+ hasTimeDimension: false,
247
+ timeFilteredShare: null,
248
+ distinctShapes: 0,
249
+ writeBurstiness: null,
250
+ errorShare: null,
251
+ missing: [],
252
+ complete: true,
253
+ };
254
+ /**
255
+ * The same features with `missing` **derived** from which values came out unknown.
256
+ *
257
+ * Derived rather than listed alongside them, so the two cannot disagree - and in the reference they
258
+ * did: `time_filtered_share` is a field neither library measures and it was left null and absent
259
+ * from this set, which breaks the one promise the set makes.
260
+ *
261
+ * What stays unknown and why, because a derivation hides the reasons. The sizes come from a storage
262
+ * sample taken in the window (`Recorder.recordStorage`, fed by `Session.measureStorage`) and are
263
+ * unknown without one. Growth needs samples at least an hour apart; burstiness needs rows written
264
+ * in the window; `time_filtered_share` needs the caller to say which fields are times.
265
+ *
266
+ * `also` carries reasons that are not field names - `no_traffic` is the only one - because a reader
267
+ * needs the difference between "this group was idle" and "this field is not measurable".
268
+ */
269
+ function withMissing(measured, also = []) {
270
+ const unknown = MEASURED_FIELDS.filter(([, property]) => measured[property] === null).map(([key]) => key);
271
+ return { ...measured, missing: [...also, ...unknown].sort(compareCodePoints) };
272
+ }
273
+ /**
274
+ * The feature vector as the document that crosses the boundary to the control plane.
275
+ *
276
+ * **A field with no value is omitted, and `missing` is what says so.** Emitting a null would work
277
+ * too - the reader treats absent and null alike - but omitting is the honest spelling of "not
278
+ * measured", and it keeps this function from having an opinion about what a null means.
279
+ */
280
+ export function featuresRecord(features) {
281
+ const body = {};
282
+ for (const [key, property] of MEASURED_FIELDS) {
283
+ const value = features[property];
284
+ if (value === null)
285
+ continue;
286
+ body[key] = value;
287
+ }
288
+ body['missing'] = [...features.missing];
289
+ body['complete'] = features.complete;
290
+ return body;
291
+ }
292
+ /** The bucket a write's rows are counted in for `write_burstiness`. */
293
+ export const SECOND_NS = 1_000_000_000;
294
+ /** The kinds of read that take a `where` and so report what they filtered on (`filtered_on`). */
295
+ export const FILTERED_KINDS = new Set(['aggregate', 'full_scan', 'range_read']);
296
+ /**
297
+ * The shortest span a daily growth is projected from.
298
+ *
299
+ * A day projected from a few seconds of writes is a number with no basis: a demonstration writing a
300
+ * hundred rows in two seconds would claim gigabytes a day. An hour is where the projection
301
+ * multiplies a measurement by 24 rather than by thousands.
302
+ */
303
+ export const GROWTH_MIN_NS = 3_600_000_000_000;
304
+ /** The unit growth is projected to, and how long the recorder keeps a group's storage samples. */
305
+ export const DAY_NS = 86_400_000_000_000;
306
+ /**
307
+ * Why a group's size stayed unknown: its engine's adapter has no catalogue to read, a table the map
308
+ * names does not exist, the catalogue refused the login (ClickHouse without the `system.parts`
309
+ * grant) or the read failed otherwise.
310
+ */
311
+ export const STORAGE_UNAVAILABLE = ['failed', 'missing_table', 'refused', 'unsupported'];
312
+ /**
313
+ * A catalogue size as an exact safe integer, or a refusal. Drivers hand a 64-bit count over as text;
314
+ * a size past 2^53 bytes would round, and a rounded size is a wrong one.
315
+ */
316
+ export function exactBytes(value) {
317
+ const parsed = typeof value === 'bigint' ? value : BigInt(String(value));
318
+ if (parsed < 0n || parsed > BigInt(Number.MAX_SAFE_INTEGER)) {
319
+ throw new Error('a storage size outside the exact integer range');
320
+ }
321
+ return Number(parsed);
322
+ }
323
+ /**
324
+ * How far behind each of this group's derived copies ran, sorted by materialisation.
325
+ *
326
+ * Read out of the histogram rather than stored, like every other percentile here. A stored
327
+ * percentile is a second copy of a fact that changes when the samples do.
328
+ */
329
+ export function windowCopies(window, group) {
330
+ return window.fanned
331
+ .filter((stats) => stats.group === group)
332
+ .sort((a, b) => compareCodePoints(a.materialization, b.materialization))
333
+ .map((stats) => ({
334
+ group: stats.group,
335
+ materialization: stats.materialization,
336
+ writes: stats.writes,
337
+ failures: stats.failures,
338
+ lagP50Ms: stats.latency.percentileMs(0.5),
339
+ lagP99Ms: stats.latency.percentileMs(0.99),
340
+ complete: stats.failures === 0,
341
+ }));
342
+ }
343
+ /**
344
+ * What each operation shape of one group measured, in the model's enumeration order.
345
+ *
346
+ * The evidence a physical design can point at. A group's features say that 62% of its calls were
347
+ * range reads; only this says *which field* they ranged over, which is the fact a key order or a
348
+ * partition is chosen from. Still no values: a shape is the structure of a call, and the fields it
349
+ * names are the model's field names (digests, when the client hashes them).
350
+ *
351
+ * The descriptor - entity, kind, fields, target - comes from the model's own enumeration by
352
+ * identifier, never from the recorder, and `target` appears only on a relation walk. A record for an
353
+ * identifier the model does not enumerate is refused rather than described by guesswork. Every
354
+ * number is an integer count or a bucket edge divided by a million, like the rest of the document.
355
+ */
356
+ export function windowShapes(window, model, group) {
357
+ const recorded = new Map(window.shapes.filter((stats) => stats.group === group).map((stats) => [stats.shapeId, stats]));
358
+ const enumerated = enumerateShapes(model).filter((shape) => recorded.has(shapeId(shape)));
359
+ const known = new Set(enumerated.map((shape) => shapeId(shape)));
360
+ const unknown = [...recorded.keys()].filter((id) => !known.has(id)).sort(compareCodePoints);
361
+ if (unknown.length > 0) {
362
+ throw new Error(`this window has records for shapes this model does not enumerate: [${unknown.join(', ')}]. ` +
363
+ `A shape is described from the model's own enumeration, never from what the recorder was ` +
364
+ `told, so a record for an identifier the model does not produce has no fields to report.`);
365
+ }
366
+ return enumerated.map((shape) => {
367
+ const stats = recorded.get(shapeId(shape));
368
+ const entry = {
369
+ id: shapeId(shape),
370
+ entity: shape.entity,
371
+ kind: shape.kind,
372
+ fields: [...shape.fields],
373
+ };
374
+ if (shape.target !== null)
375
+ entry['target'] = shape.target;
376
+ entry['calls'] = stats.calls;
377
+ entry['errors'] = stats.errors;
378
+ entry['rows'] = stats.rows;
379
+ entry['latency_p50_ms'] = stats.latency.percentileMs(0.5);
380
+ entry['latency_p99_ms'] = stats.latency.percentileMs(0.99);
381
+ if (stats.filtered.size > 0) {
382
+ // What the calls filtered on: one entry per combination of equality fields and bounded
383
+ // field, in code point order. No `equal` fields and no `range` is the calls that filtered
384
+ // on nothing.
385
+ entry['filtered_on'] = [...stats.filtered.values()]
386
+ .sort((a, b) => comparePredicates(a.predicates, b.predicates))
387
+ .map(({ predicates, calls }) => ({
388
+ equal: [...predicates.equal],
389
+ ...(predicates.ranged === '' ? {} : { range: predicates.ranged }),
390
+ calls,
391
+ }));
392
+ }
393
+ return entry;
394
+ });
395
+ }
396
+ /**
397
+ * Nearest-rank: the smallest sample at least `fraction` of the data is not above.
398
+ *
399
+ * **The same rank rule as `Histogram.percentileMs`, and it did not used to be.** This took the
400
+ * floor while the histogram takes the first bucket whose cumulative count reaches
401
+ * `fraction * count`, which is the ceiling. The two agree on every sample set with an odd count,
402
+ * and every case in `telemetry/` had one, so one window document carried two percentile
403
+ * conventions and nothing could see it. They differ exactly when `fraction * length` is an
404
+ * integer, which is what two samples at p50 is: `[3, 300]` reported 300 here and the *lower*
405
+ * bucket in the histogram. `telemetry/007` pins it.
406
+ */
407
+ function at(ordered, fraction) {
408
+ if (ordered.length === 0)
409
+ return null;
410
+ const rank = Math.max(1, Math.ceil(fraction * ordered.length)) - 1;
411
+ return ordered[Math.min(ordered.length - 1, rank)] ?? null;
412
+ }
413
+ /** Whole seconds in a non-negative integer count of nanoseconds, in integer arithmetic. */
414
+ function wholeSeconds(nanoseconds) {
415
+ return (nanoseconds - (nanoseconds % SECOND_NS)) / SECOND_NS;
416
+ }
417
+ /**
418
+ * `total_bytes`, `index_to_table_ratio` and `daily_growth_bytes` from the window's samples.
419
+ *
420
+ * The size is the latest sample taken **in this window**. Growth reaches back to the oldest sample
421
+ * kept (up to a day), is projected to a day only from a span of at least `GROWTH_MIN_NS`, truncated
422
+ * toward zero - in `bigint`, because bytes times a day in nanoseconds passes 2^53 - and may be
423
+ * negative: ClickHouse merges shrink parts.
424
+ */
425
+ function storageFeatures(window, group) {
426
+ let latest;
427
+ for (const sample of window.storage) {
428
+ if (sample.group === group && (latest === undefined || sample.atNs >= latest.atNs))
429
+ latest = sample;
430
+ }
431
+ if (latest === undefined)
432
+ return [null, null, null];
433
+ const rest = latest.totalBytes - latest.secondaryIndexBytes;
434
+ const ratio = rest > 0 ? latest.secondaryIndexBytes / rest : null;
435
+ let oldest = latest;
436
+ let first = true;
437
+ for (const sample of window.storageHistory) {
438
+ if (sample.group !== group)
439
+ continue;
440
+ if (first || sample.atNs < oldest.atNs)
441
+ oldest = sample;
442
+ first = false;
443
+ }
444
+ const elapsed = latest.atNs - oldest.atNs;
445
+ let growth = null;
446
+ if (elapsed >= GROWTH_MIN_NS) {
447
+ const delta = latest.totalBytes - oldest.totalBytes;
448
+ const projected = (BigInt(Math.abs(delta)) * BigInt(DAY_NS)) / BigInt(Math.trunc(elapsed));
449
+ growth = Number(delta >= 0 ? projected : -projected);
450
+ }
451
+ return [latest.totalBytes, ratio, growth];
452
+ }
453
+ /**
454
+ * How many times the busiest second's rows exceed the window's mean rate: `M * S / W`, with `S` the
455
+ * window's whole seconds - at least one, and at least as many as the seconds its writes landed in.
456
+ */
457
+ function burstiness(window, group) {
458
+ const seconds = window.writeSeconds.get(group);
459
+ if (seconds === undefined)
460
+ return null;
461
+ let written = 0;
462
+ let busiest = 0;
463
+ let last = 0;
464
+ for (const [second, rows] of seconds) {
465
+ written += rows;
466
+ busiest = Math.max(busiest, rows);
467
+ last = Math.max(last, second);
468
+ }
469
+ if (written <= 0)
470
+ return null;
471
+ const duration = window.endedNs - window.startedNs;
472
+ const ceiling = duration > 0 ? wholeSeconds(duration) + (duration % SECOND_NS > 0 ? 1 : 0) : 0;
473
+ const span = Math.max(1, ceiling, last + 1);
474
+ return (busiest * span) / written;
475
+ }
476
+ /** Fold this window's records for one group into the planner's feature vector. */
477
+ export function windowFeatures(window, group, options = {}) {
478
+ const timeDimension = options.hasTimeDimension === true;
479
+ const records = window.shapes.filter((stats) => stats.group === group);
480
+ if (records.length === 0) {
481
+ // `no_traffic` is the *reason*, and the unknown fields are named too, by the same derivation as
482
+ // below - two branches of one function computing `missing` by different rules is the defect
483
+ // this set exists to prevent.
484
+ return withMissing({ ...EMPTY_FEATURES, complete: window.complete }, ['no_traffic']);
485
+ }
486
+ let writes = 0;
487
+ let reads = 0;
488
+ for (const stats of records) {
489
+ if (WRITE_KINDS.has(stats.kind))
490
+ writes += stats.calls;
491
+ else
492
+ reads += stats.calls;
493
+ }
494
+ const calls = writes + reads;
495
+ const latency = new Histogram();
496
+ for (const stats of records)
497
+ latency.merge(stats.latency);
498
+ const counts = new Map();
499
+ for (const stats of records)
500
+ counts.set(stats.kind, (counts.get(stats.kind) ?? 0) + stats.calls);
501
+ const shapeMix = {};
502
+ if (calls > 0) {
503
+ for (const kind of [...counts.keys()].sort(compareCodePoints)) {
504
+ shapeMix[kind] = (counts.get(kind) ?? 0) / calls;
505
+ }
506
+ }
507
+ // A failed call counts in the latency histogram and **not** here, and the two answers have
508
+ // different reasons rather than one convention. A failure took time, so dropping it from the
509
+ // histogram would flatter a window precisely when the engine is in trouble. It returned no rows
510
+ // because it failed rather than because the data is sparse, so averaging that zero in
511
+ // understates how many rows a read of this shape returns - and `errorShare` already carries the
512
+ // failure rate, so smearing it into a second feature is one fact in two places. A shape whose
513
+ // every call failed contributes nothing rather than a zero. `telemetry/008`.
514
+ const readRecords = records.filter((stats) => !WRITE_KINDS.has(stats.kind) && stats.calls > stats.errors);
515
+ // A numeric comparator, because the default one sorts lexicographically and would put 10 before
516
+ // 2. `telemetry/003` has cardinalities that expose it.
517
+ const cardinalities = readRecords
518
+ .map((stats) => stats.rows / (stats.calls - stats.errors))
519
+ .sort((a, b) => a - b);
520
+ const pkCalls = records
521
+ .filter((stats) => stats.kind === 'point_read')
522
+ .reduce((sum, stats) => sum + stats.calls, 0);
523
+ const errors = records.reduce((sum, stats) => sum + stats.errors, 0);
524
+ // A call filtered on time when its filter bounded a time field by a range or compared one by
525
+ // equality. Names only - the fields `filtered_on` already reports. The denominator is every call
526
+ // of the group, as for `pk_access_share`.
527
+ // A read that takes a `where` and did not report its filters leaves the share unknown rather than
528
+ // counting it as not filtered on time - except a range read, whose shape names the field it
529
+ // bounded; in an entity with no time field nothing could have filtered on time.
530
+ let timeFilteredShare = null;
531
+ if (options.timeFields !== undefined && calls > 0) {
532
+ let filtered = 0;
533
+ let known = true;
534
+ for (const stats of records) {
535
+ const times = options.timeFields.get(stats.entity);
536
+ if (times === undefined || times.size === 0)
537
+ continue;
538
+ let reported = 0;
539
+ for (const { predicates, calls: hits } of stats.filtered.values()) {
540
+ reported += hits;
541
+ if (times.has(predicates.ranged) || predicates.equal.some((name) => times.has(name))) {
542
+ filtered += hits;
543
+ }
544
+ }
545
+ const unreported = stats.calls - reported;
546
+ if (unreported <= 0 || !FILTERED_KINDS.has(stats.kind))
547
+ continue;
548
+ const bounded = options.rangeFields?.get(stats.shapeId);
549
+ if (stats.kind === 'range_read' && bounded !== undefined) {
550
+ if (times.has(bounded))
551
+ filtered += unreported;
552
+ }
553
+ else {
554
+ known = false;
555
+ }
556
+ }
557
+ timeFilteredShare = known ? filtered / calls : null;
558
+ }
559
+ const [totalBytes, indexToTableRatio, dailyGrowthBytes] = storageFeatures(window, group);
560
+ return withMissing({
561
+ ...EMPTY_FEATURES,
562
+ calls,
563
+ readWriteRatio: writes > 0 ? reads / writes : null,
564
+ shapeMix,
565
+ latencyP50Ms: latency.percentileMs(0.5),
566
+ latencyP99Ms: latency.percentileMs(0.99),
567
+ resultCardinalityP50: at(cardinalities, 0.5),
568
+ resultCardinalityP99: at(cardinalities, 0.99),
569
+ totalBytes,
570
+ dailyGrowthBytes,
571
+ indexToTableRatio,
572
+ pkAccessShare: calls > 0 ? pkCalls / calls : null,
573
+ hasTimeDimension: timeDimension,
574
+ timeFilteredShare,
575
+ distinctShapes: records.length,
576
+ writeBurstiness: burstiness(window, group),
577
+ errorShare: calls > 0 ? errors / calls : null,
578
+ complete: window.complete,
579
+ });
580
+ }
581
+ /**
582
+ * This window as the document the control plane reads. Numbers, never rows.
583
+ *
584
+ * The model is required and it is checked. Only one fact is read from it - whether a group carries a
585
+ * time dimension, which is decided by declared *type* and never by a field's name - but a window
586
+ * serialised against the wrong model would attach that fact to the wrong groups and claim
587
+ * `has_time_dimension: false` for a group that has one. False is a claim; a measurement this
588
+ * library cannot make has to be absent.
589
+ *
590
+ * **This document is deliberately not canonical, and that needs saying because section 1 of the
591
+ * format contract rejects floating point outright.** Its reason is that a float's textual form
592
+ * differs between languages, and almost every number here is a float. This document is not signed,
593
+ * not hashed and never compared for equality, so the rule it breaks does not apply - but the thing
594
+ * that makes the `telemetry/` family checkable is narrower and worth stating: every number here is
595
+ * either **a ratio of two integers** or **a bucket edge divided by a million**, and IEEE 754
596
+ * requires division to be correctly rounded. So two languages compute the same double from the same
597
+ * traffic even where they would print it differently, which is why those vectors compare numbers
598
+ * rather than bytes.
599
+ */
600
+ export function windowRecord(window, model) {
601
+ if (model.version !== window.modelVersion) {
602
+ throw new Error(`this window measured model version ${window.modelVersion} and it is being serialised ` +
603
+ `against ${model.version}. Only one fact is read from the model here - whether a group ` +
604
+ `has a time dimension - and reading it from another model would attach it to the wrong ` +
605
+ `groups. Keep the model the recorder was created for, or drop the window: a window ` +
606
+ `measured against a model that no longer exists cannot be scored against the one that ` +
607
+ `does.`);
608
+ }
609
+ const byName = new Map(colocationGroups(model).map((group) => [group.name, group]));
610
+ const named = [
611
+ ...new Set([
612
+ ...window.shapes.map((stats) => stats.group),
613
+ ...window.fanned.map((stats) => stats.group),
614
+ ]),
615
+ ].sort(compareCodePoints);
616
+ const unknown = named.filter((name) => !byName.has(name));
617
+ if (unknown.length > 0) {
618
+ throw new Error(`this window has records for groups this model does not have: [${unknown.join(', ')}]. The ` +
619
+ `recorder is given a model version and the groups come from the operations it observed, ` +
620
+ `so this is a session routing a model other than the one measured.`);
621
+ }
622
+ const rangeFields = new Map(enumerateShapes(model)
623
+ .filter((shape) => shape.kind === 'range_read' && shape.fields.length > 0)
624
+ .map((shape) => [shapeId(shape), shape.fields[0]]));
625
+ const groups = {};
626
+ for (const name of named) {
627
+ const group = byName.get(name);
628
+ if (group === undefined)
629
+ continue;
630
+ const record = featuresRecord(windowFeatures(window, name, {
631
+ hasTimeDimension: hasTimeDimension(model, group),
632
+ timeFields: timeFields(model, group),
633
+ rangeFields,
634
+ }));
635
+ const shapes = windowShapes(window, model, name);
636
+ if (shapes.length > 0) {
637
+ // Absent rather than empty, like `copies`: a group in this document saw traffic, so an empty
638
+ // list would only arise from a group listed for its copies alone.
639
+ record['shapes'] = shapes;
640
+ }
641
+ const copies = windowCopies(window, name).map(copyFreshnessRecord);
642
+ if (copies.length > 0) {
643
+ // Absent rather than empty when the group has no derived copy, for the reason `also_write` is
644
+ // absent rather than empty in a map: absent says "not doing this" and an empty list says
645
+ // "considered and found none", which is a stronger claim.
646
+ record['copies'] = copies;
647
+ }
648
+ groups[name] = record;
649
+ }
650
+ return {
651
+ model_version: window.modelVersion,
652
+ complete: window.complete,
653
+ dropped_windows: window.droppedWindows,
654
+ groups,
655
+ };
656
+ }
657
+ /** Types that make a field a time dimension. Recognised by **type**, never by name. */
658
+ const TIME_TYPES = new Set(['date', 'timestamp', 'timestamptz']);
659
+ /**
660
+ * Does any entity in this group carry a time dimension?
661
+ *
662
+ * Decided by the declared type and never by the field's name. A name is not evidence: `created_at`
663
+ * typed as a string is a string, and treating it as a timestamp would have the planner recommend
664
+ * time partitioning on a column no engine can range-scan usefully. And a client may hash identifier
665
+ * names so that we never see them - a derivation that read names would silently produce different
666
+ * answers with hashing on and off, which is the property the hashed-model vectors forbid.
667
+ */
668
+ export function hasTimeDimension(model, group) {
669
+ for (const member of group.members) {
670
+ const spec = model.entities.find((entity) => entity.name === member);
671
+ if (spec === undefined)
672
+ continue;
673
+ if (spec.fields.some((field) => TIME_TYPES.has(field.type)))
674
+ return true;
675
+ }
676
+ return false;
677
+ }
678
+ /**
679
+ * Each entity of this group and its fields of a time type - by type, never by name.
680
+ *
681
+ * What `time_filtered_share` counts against, for the reasons `hasTimeDimension` gives: a
682
+ * `created_at` typed as a string is not a time, and a hashed model must answer the same.
683
+ */
684
+ export function timeFields(model, group) {
685
+ const out = new Map();
686
+ for (const member of group.members) {
687
+ const spec = model.entities.find((entity) => entity.name === member);
688
+ out.set(member, new Set((spec?.fields ?? []).filter((field) => TIME_TYPES.has(field.type)).map((f) => f.name)));
689
+ }
690
+ return out;
691
+ }
692
+ /**
693
+ * Accumulates records, rolls windows, and drops telemetry rather than anything else.
694
+ *
695
+ * No lock, because this runtime does not need one - and that is worth stating rather than leaving
696
+ * as an absence. The reference implementation takes one only when a shape is first seen or a window
697
+ * rolls, and accepts that two threads racing on the same shape can lose a call from a count. Here
698
+ * there is nothing to race: the cost this class is allowed is zero either way.
699
+ */
700
+ export class Recorder {
701
+ modelVersion;
702
+ maxWindows;
703
+ clock;
704
+ current = new Map();
705
+ fanned = new Map();
706
+ writes = new Map();
707
+ storage = new Map();
708
+ storageWindow = [];
709
+ startedNs;
710
+ windows = [];
711
+ dropped = 0;
712
+ incomplete = false;
713
+ /**
714
+ * `clock` returns nanoseconds and defaults to a monotonic one; tests and the conformance vectors
715
+ * pass their own, because a write's second and a sample's age are read from it.
716
+ */
717
+ constructor(modelVersion, maxWindows = 64, clock = now) {
718
+ this.modelVersion = modelVersion;
719
+ this.maxWindows = maxWindows;
720
+ this.clock = clock;
721
+ this.startedNs = this.clock();
722
+ }
723
+ /** Record one operation. Never throws. */
724
+ record(options) {
725
+ guard('telemetry.record', () => {
726
+ let stats = this.current.get(options.shapeId);
727
+ if (stats === undefined) {
728
+ stats = new ShapeStats(options.shapeId, options.group, options.entity, options.kind);
729
+ this.current.set(options.shapeId, stats);
730
+ }
731
+ const predicates = options.equal === undefined
732
+ ? undefined
733
+ : {
734
+ equal: [...new Set(options.equal)].sort(compareCodePoints),
735
+ ranged: options.ranged ?? '',
736
+ };
737
+ const rows = options.rows ?? 0;
738
+ stats.record(options.nanoseconds, rows, options.failed === true, predicates);
739
+ if (rows > 0 && options.failed !== true && WRITE_KINDS.has(options.kind)) {
740
+ // The second of the window this write's rows landed in: one clock read per write.
741
+ const second = wholeSeconds(Math.max(0, this.clock() - this.startedNs));
742
+ let seconds = this.writes.get(options.group);
743
+ if (seconds === undefined) {
744
+ seconds = new Map();
745
+ this.writes.set(options.group, seconds);
746
+ }
747
+ seconds.set(second, (seconds.get(second) ?? 0) + rows);
748
+ }
749
+ });
750
+ }
751
+ /**
752
+ * Record one group's size, as the engine's catalogue reported it. Never throws.
753
+ *
754
+ * `Session.measureStorage` calls this for every group it could measure. A sample is kept for a
755
+ * day, across windows, because growth is a property of time rather than of one window; a sample
756
+ * that is not two non-negative safe integers with the index part inside the total is dropped,
757
+ * never guessed into shape.
758
+ */
759
+ recordStorage(options) {
760
+ guard('telemetry.record_storage', () => {
761
+ const { group, totalBytes, secondaryIndexBytes } = options;
762
+ if (typeof group !== 'string' ||
763
+ group.length === 0 ||
764
+ !Number.isSafeInteger(totalBytes) ||
765
+ !Number.isSafeInteger(secondaryIndexBytes) ||
766
+ totalBytes < 0 ||
767
+ secondaryIndexBytes < 0 ||
768
+ secondaryIndexBytes > totalBytes) {
769
+ return;
770
+ }
771
+ const sample = { group, atNs: this.clock(), totalBytes, secondaryIndexBytes };
772
+ const kept = (this.storage.get(group) ?? []).filter((s) => s.atNs >= sample.atNs - DAY_NS);
773
+ kept.push(sample);
774
+ this.storage.set(group, kept);
775
+ this.storageWindow.push(sample);
776
+ });
777
+ }
778
+ /**
779
+ * Record one write to one derived copy. Never throws.
780
+ *
781
+ * A separate entry point from `record` rather than a shape kind, because a fan-out is not an
782
+ * operation the application asked for - see `FanOutStats`.
783
+ */
784
+ recordFanOut(options) {
785
+ guard('telemetry.record_fan_out', () => {
786
+ const key = `${options.group} ${options.materialization}`;
787
+ let stats = this.fanned.get(key);
788
+ if (stats === undefined) {
789
+ stats = new FanOutStats(options.group, options.materialization);
790
+ this.fanned.set(key, stats);
791
+ }
792
+ stats.record(options.nanoseconds, options.failed === true);
793
+ });
794
+ }
795
+ /**
796
+ * Close the current period and queue it.
797
+ *
798
+ * Returns the window, or undefined if nothing was recorded in it.
799
+ */
800
+ roll() {
801
+ return guard('telemetry.roll', () => {
802
+ if (this.current.size === 0 && this.fanned.size === 0) {
803
+ // The fan-out half is not redundant. A window holding only fan-out records cannot arise
804
+ // from an application write - a write records a shape and then fans out - but it can from a
805
+ // backfill replaying rows, and a window silently discarded is the shape of gap that makes a
806
+ // client's copy look healthier than it is.
807
+ return undefined;
808
+ }
809
+ const window = {
810
+ modelVersion: this.modelVersion,
811
+ startedNs: this.startedNs,
812
+ endedNs: this.clock(),
813
+ shapes: [...this.current.values()],
814
+ complete: !this.incomplete,
815
+ droppedWindows: this.dropped,
816
+ fanned: [...this.fanned.values()],
817
+ storage: [...this.storageWindow],
818
+ storageHistory: [...this.storage.values()].flat(),
819
+ writeSeconds: new Map([...this.writes].map(([group, seconds]) => [group, new Map(seconds)])),
820
+ };
821
+ this.current = new Map();
822
+ this.fanned = new Map();
823
+ this.writes = new Map();
824
+ this.storageWindow = [];
825
+ this.startedNs = this.clock();
826
+ this.incomplete = false;
827
+ // A full buffer drops the oldest window and says so in the next one. Telemetry is the thing
828
+ // that gets lost when we run out of room - never a write, never an operation.
829
+ if (this.windows.length === this.maxWindows) {
830
+ this.windows.shift();
831
+ this.dropped += 1;
832
+ }
833
+ this.windows.push(window);
834
+ return window;
835
+ });
836
+ }
837
+ pending() {
838
+ return [...this.windows];
839
+ }
840
+ /**
841
+ * Called by the application when part of the current period was not recorded.
842
+ *
843
+ * The flag travels in `Window.complete` and the planner reads it, because a window missing a
844
+ * slice of the traffic must not be scored as though it were the whole period. Nothing here
845
+ * decides when that happened: the application collects these windows and hands them over, and
846
+ * only it knows whether a collection was skipped.
847
+ */
848
+ markIncomplete() {
849
+ this.incomplete = true;
850
+ }
851
+ /**
852
+ * Drop the oldest `count` windows once the application has taken them.
853
+ *
854
+ * The delivering party is the client's own process, and there is no other. This library never
855
+ * opens a connection to us - the buffer is read with `pending`, written wherever the application
856
+ * writes it, and that file is what we are handed.
857
+ */
858
+ acknowledge(count) {
859
+ this.windows = this.windows.slice(Math.min(count, this.windows.length));
860
+ }
861
+ }
862
+ /**
863
+ * A monotonic clock in nanoseconds.
864
+ *
865
+ * `process.hrtime.bigint()` rather than `Date.now()`: a window's start and end are used to compute
866
+ * durations, and a wall clock can move backwards. Converted to a number because everything that
867
+ * consumes it is arithmetic on durations well inside the safe integer range - 2^53 ns is 104 days.
868
+ */
869
+ function now() {
870
+ return Number(process.hrtime.bigint());
871
+ }
872
+ //# sourceMappingURL=telemetry.js.map