@forge-ops/tracker 0.5.0 → 0.9.0

This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
@@ -0,0 +1,117 @@
1
+ /**
2
+ * Collects individual captureMetric()/captureInfrastructureMetric() calls in-process and periodically
3
+ * flushes them as one batch, rather than one network call per capture. Unlike performanceFlusher.js
4
+ * this keeps a *list* of individually meaningful entries instead of summing them into buckets: a
5
+ * customer's own signup or payment is exactly the kind of thing they will want a genuinely accurate
6
+ * count/sum of later, so the server stores one row per entry as-is. Ported from
7
+ * gems/forge_ops_tracker/lib/forge_ops_tracker/metric_buffer.rb and infrastructure_metric_buffer.rb,
8
+ * which are the same class twice; here it is one class instantiated twice, told which delivery
9
+ * method and flush interval to use.
10
+ *
11
+ * Three deliberate differences from the Ruby buffers:
12
+ *
13
+ * - A flush snapshots the first N entries and, on success, removes exactly those N, instead of
14
+ * resetting the whole list: an entry recorded while the request is in flight (this is async, so
15
+ * that window is real) is kept for the next flush rather than lost.
16
+ * - The buffer is capped at MAX_ENTRIES, and once full further entries are dropped until a flush
17
+ * succeeds: a plan without the feature answers 403 on every flush, and an uncapped buffer would
18
+ * then grow for as long as the process lives. Dropping the newest rather than the oldest keeps
19
+ * the entries a flush is delivering at the front of the list, which is what makes removing
20
+ * exactly those afterward exact.
21
+ * - A NaN or infinite value is dropped at record time: JSON.stringify turns it into `null`, which
22
+ * the server would reject, taking the whole batch with it.
23
+ *
24
+ * setInterval(...).unref(), not a real background thread, same reasoning sessionFlusher.js's own
25
+ * header comment documents. Unlike that flusher this also flushes on "beforeExit" (fired when the
26
+ * event loop has nothing left, and which does not change how the process exits the way a
27
+ * SIGTERM/SIGINT listener would): the real use of the infrastructure endpoint is a short-lived cron
28
+ * script that captures a few readings and lets the process end.
29
+ */
30
+ export const MAX_ENTRIES = 1000;
31
+
32
+ export class MetricBuffer {
33
+ #configuration;
34
+ #deliver;
35
+ #intervalMs;
36
+ #entries = [];
37
+ #timer = null;
38
+ #beforeExit = null;
39
+
40
+ /**
41
+ * @param {import("./configuration.js").Configuration} configuration
42
+ * @param {(entries: Record<string, unknown>[]) => Promise<boolean>} deliver
43
+ * @param {() => number} intervalMs read fresh each time the timer is created
44
+ */
45
+ constructor(configuration, deliver, intervalMs) {
46
+ this.#configuration = configuration;
47
+ this.#deliver = deliver;
48
+ this.#intervalMs = intervalMs;
49
+ }
50
+
51
+ /**
52
+ * @param {Record<string, unknown>} entry everything but recorded_at
53
+ * @returns {boolean} whether it was kept
54
+ */
55
+ record(entry) {
56
+ const value = entry.value;
57
+ if (typeof value !== "number" || !Number.isFinite(value)) {
58
+ this.#configuration.log("[forge-ops-tracker] dropped a metric with a non-numeric or non-finite value");
59
+ return false;
60
+ }
61
+ if (this.#entries.length >= MAX_ENTRIES) {
62
+ this.#configuration.log("[forge-ops-tracker] metric buffer full, dropping a metric");
63
+ return false;
64
+ }
65
+
66
+ this.#ensureStarted();
67
+ this.#entries.push({ ...entry, recorded_at: `${new Date().toISOString().slice(0, 19)}Z` });
68
+ return true;
69
+ }
70
+
71
+ /** Delivers everything buffered so far as one batch. A failed delivery keeps every entry, so the next flush's batch just grows. */
72
+ async flush() {
73
+ if (this.#entries.length === 0) {
74
+ return;
75
+ }
76
+
77
+ const snapshot = this.#entries.slice();
78
+ const delivered = await this.#deliver(snapshot);
79
+ if (!delivered) {
80
+ return;
81
+ }
82
+
83
+ // Exactly the entries just delivered: anything recorded while the request was in flight sits
84
+ // after them and stays for the next flush.
85
+ this.#entries.splice(0, snapshot.length);
86
+ }
87
+
88
+ /** Stops the timer and the exit hook without delivering anything. */
89
+ discard() {
90
+ if (this.#timer !== null) {
91
+ clearInterval(this.#timer);
92
+ this.#timer = null;
93
+ }
94
+ if (this.#beforeExit !== null) {
95
+ process.off("beforeExit", this.#beforeExit);
96
+ this.#beforeExit = null;
97
+ }
98
+ this.#entries = [];
99
+ }
100
+
101
+ #ensureStarted() {
102
+ if (this.#timer !== null) {
103
+ return;
104
+ }
105
+
106
+ this.#timer = setInterval(() => this.#flushSafely(), this.#intervalMs());
107
+ this.#timer.unref();
108
+ this.#beforeExit = () => this.#flushSafely();
109
+ process.on("beforeExit", this.#beforeExit);
110
+ }
111
+
112
+ #flushSafely() {
113
+ return this.flush().catch((e) => {
114
+ this.#configuration.log(`[forge-ops-tracker] metric flush error: ${e.name}: ${e.message}`);
115
+ });
116
+ }
117
+ }
@@ -1,3 +1,5 @@
1
+ import { bucketFor } from "./histogramBucketer.js";
2
+
1
3
  /**
2
4
  * Times requests in-process, bucketed by transactionName (see integrations/performance.js), and
3
5
  * periodically flushes each distinct bucket as one small aggregate report, rather than one
@@ -33,19 +35,30 @@ export class PerformanceFlusher {
33
35
  record(transactionName, durationMs) {
34
36
  this.#ensureTimerStarted();
35
37
 
36
- const bucket = this.#buckets.get(transactionName) ?? { count: 0, durationSumMs: 0, maxDurationMs: 0 };
38
+ const bucket = this.#buckets.get(transactionName) ?? { count: 0, durationSumMs: 0, maxDurationMs: 0, histogram: {} };
37
39
  bucket.count += 1;
38
40
  bucket.durationSumMs += durationMs;
39
41
  if (durationMs > bucket.maxDurationMs) {
40
42
  bucket.maxDurationMs = durationMs;
41
43
  }
44
+ // The distribution count/sum/max can't reconstruct: see histogramBucketer.js for why the
45
+ // server approximates a percentile from these bucket counts.
46
+ const label = bucketFor(durationMs);
47
+ bucket.histogram[label] = (bucket.histogram[label] ?? 0) + 1;
42
48
  this.#buckets.set(transactionName, bucket);
43
49
  }
44
50
 
45
51
  /**
46
- * Snapshots and resets the buckets, then delivers them as one batch. A failed delivery keeps
47
- * every bucket where it is rather than resetting, so the next flush's batch just grows instead
48
- * of losing what was already tallied; same reasoning sessionFlusher.js's own flush() documents.
52
+ * Snapshots the buckets, then delivers them as one batch. A failed delivery keeps every bucket
53
+ * where it is rather than resetting, so the next flush's batch just grows instead of losing what
54
+ * was already tallied; same reasoning sessionFlusher.js's own flush() documents.
55
+ *
56
+ * Only exactly what this snapshot delivered is removed afterward, subtracted from whatever is in
57
+ * each bucket by then, never the whole map reset: record() can run while the delivery is awaited,
58
+ * so a record for a transaction already in the snapshot, or a brand-new one, can land between the
59
+ * snapshot and delivery succeeding, and resetting afterward would silently discard it.
60
+ * maxDurationMs is left as whatever is currently on the bucket, sent or not: a max can't be
61
+ * "subtracted" back out, and leaving it never overstates the next period's own max.
49
62
  */
50
63
  async flush() {
51
64
  if (this.#buckets.size === 0) {
@@ -53,6 +66,12 @@ export class PerformanceFlusher {
53
66
  }
54
67
 
55
68
  const periodEndedAt = new Date();
69
+ const sent = new Map(
70
+ [...this.#buckets.entries()].map(([transactionName, bucket]) => [
71
+ transactionName,
72
+ { count: bucket.count, durationSumMs: bucket.durationSumMs, histogram: { ...bucket.histogram } },
73
+ ]),
74
+ );
56
75
  const samples = [...this.#buckets.entries()].map(([transactionName, bucket]) => ({
57
76
  transaction_name: transactionName,
58
77
  environment: this.#configuration.environment,
@@ -62,6 +81,7 @@ export class PerformanceFlusher {
62
81
  request_count: bucket.count,
63
82
  duration_sum_ms: bucket.durationSumMs,
64
83
  max_duration_ms: bucket.maxDurationMs,
84
+ histogram: { ...bucket.histogram },
65
85
  }));
66
86
 
67
87
  const delivered = await this.#client.deliverPerformanceSamples(samples);
@@ -69,7 +89,25 @@ export class PerformanceFlusher {
69
89
  return;
70
90
  }
71
91
 
72
- this.#buckets = new Map();
92
+ for (const [transactionName, deliveredBucket] of sent) {
93
+ const current = this.#buckets.get(transactionName);
94
+ if (!current) {
95
+ continue;
96
+ }
97
+ current.count -= deliveredBucket.count;
98
+ current.durationSumMs = Math.max(current.durationSumMs - deliveredBucket.durationSumMs, 0);
99
+ for (const [label, count] of Object.entries(deliveredBucket.histogram)) {
100
+ const remaining = (current.histogram[label] ?? 0) - count;
101
+ if (remaining > 0) {
102
+ current.histogram[label] = remaining;
103
+ } else {
104
+ delete current.histogram[label];
105
+ }
106
+ }
107
+ if (current.count <= 0) {
108
+ this.#buckets.delete(transactionName);
109
+ }
110
+ }
73
111
  this.#periodStartedAt = periodEndedAt;
74
112
  }
75
113
 
package/src/reporter.js CHANGED
@@ -21,14 +21,15 @@ export class Reporter {
21
21
  * @param {Error} error
22
22
  * @param {Record<string, unknown>} [context]
23
23
  * @param {Record<string, unknown> | null} [user]
24
+ * @param {Array<Record<string, unknown>>} [breadcrumbs]
24
25
  */
25
- report(error, context = {}, user = null) {
26
+ report(error, context = {}, user = null, breadcrumbs = []) {
26
27
  try {
27
28
  if (!this.#configuration.isEnabled()) {
28
29
  return;
29
30
  }
30
31
 
31
- const payload = this.#eventBuilder.build(error, context, user);
32
+ const payload = this.#eventBuilder.build(error, context, user, breadcrumbs);
32
33
  this.#deliveryQueue.push(payload);
33
34
  } catch (e) {
34
35
  this.#configuration.log(`[forge-ops-tracker] report failed: ${e.name}: ${e.message}`);
@@ -0,0 +1,89 @@
1
+ import { randomBytes } from "node:crypto";
2
+
3
+ /**
4
+ * A short, unique-enough identifier for one span: 8 random bytes as hex, matching
5
+ * gems/forge_ops_tracker/lib/forge_ops_tracker/span_buffer.rb's own SecureRandom.hex(8). The
6
+ * server only ever needs these to be unique within one trace's own array (see
7
+ * Api::V1::SpansController on the Rails side), never a real database id, so this is deliberately
8
+ * cheap rather than a full UUID.
9
+ * @returns {string}
10
+ */
11
+ export function randomSpanId() {
12
+ return randomBytes(8).toString("hex");
13
+ }
14
+
15
+ /**
16
+ * Accumulates one request's own nested call tree in-process: the tracing analog to
17
+ * BreadcrumbBuffer's per-request trail. Ported from
18
+ * gems/forge_ops_tracker/lib/forge_ops_tracker/span_buffer.rb, with one deliberate departure: the
19
+ * Ruby buffer tracks "what's currently open" with a single mutable stack, which is correct for
20
+ * Ruby's own single-threaded-per-request execution model, but is the wrong tool here. A customer
21
+ * can easily await two service-layer span() calls concurrently in the same request (Promise.all
22
+ * rather than one after another); popping one call's own span off a shared stack while the other
23
+ * is still pending would leave the second one mis-parented under the first instead of under the
24
+ * request's own root. index.js's own span()/leaf-span recording instead tracks "the currently
25
+ * open span" with a second, nested AsyncLocalStorage (see spanParentStorage there): each
26
+ * concurrent branch gets its own correctly-scoped view automatically, the exact property that
27
+ * storage exists to provide. This class only ever holds what doesn't depend on that: the trace's
28
+ * own identifiers and the flat list every span, root or leaf, ends up recorded into.
29
+ */
30
+ export class SpanBuffer {
31
+ #configuration;
32
+ traceId;
33
+ rootSpanId;
34
+ spans = [];
35
+ #rootDurationMs = null;
36
+
37
+ constructor(configuration) {
38
+ this.#configuration = configuration;
39
+ this.traceId = randomBytes(16).toString("hex");
40
+ this.rootSpanId = randomSpanId();
41
+ }
42
+
43
+ /**
44
+ * root: true only for the one call that records the request's own root span: forces
45
+ * parent_span_id null explicitly, the same reasoning span_buffer.rb's own #record documents for
46
+ * why that can't just be read off of "whatever's currently open" for the root specifically.
47
+ *
48
+ * @param {{
49
+ * spanId: string,
50
+ * parentSpanId?: string | null,
51
+ * name: string,
52
+ * kind: string,
53
+ * startedAt: Date,
54
+ * durationMs: number,
55
+ * data?: Record<string, unknown>,
56
+ * root?: boolean,
57
+ * }} span
58
+ */
59
+ record({ spanId, parentSpanId = null, name, kind, startedAt, durationMs, data = {}, root = false }) {
60
+ this.spans.push({
61
+ span_id: spanId,
62
+ parent_span_id: root ? null : parentSpanId,
63
+ name,
64
+ kind,
65
+ // Date#toISOString() already yields millisecond precision UTC ("...sssZ"), exactly the wire
66
+ // format the server expects; unlike BreadcrumbBuffer's own timestamp, this one must NOT
67
+ // strip milliseconds, since a waterfall's own ordering depends on them.
68
+ started_at: startedAt.toISOString(),
69
+ duration_ms: durationMs,
70
+ environment: this.#configuration.environment,
71
+ release: this.#configuration.release,
72
+ data,
73
+ });
74
+ if (root) {
75
+ this.#rootDurationMs = durationMs;
76
+ }
77
+ }
78
+
79
+ /**
80
+ * null (never sent) until the root span has actually been recorded: a request whose own
81
+ * tracing integration never got the chance to call back at all has no duration to compare
82
+ * against a threshold, so it's never mistakenly treated as slow.
83
+ * @param {number} thresholdMs
84
+ * @returns {boolean}
85
+ */
86
+ isSlow(thresholdMs) {
87
+ return this.#rootDurationMs !== null && this.#rootDurationMs >= thresholdMs;
88
+ }
89
+ }