@forge-ops/tracker 0.5.0 → 0.9.0
This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
- package/README.md +161 -2
- package/package.json +4 -2
- package/src/breadcrumbBuffer.js +48 -0
- package/src/client.js +29 -0
- package/src/configuration.js +73 -0
- package/src/deliveryQueue.js +16 -3
- package/src/eventBuilder.js +15 -3
- package/src/histogramBucketer.js +26 -0
- package/src/httpTracing.js +104 -0
- package/src/index.js +371 -1
- package/src/integrations/breadcrumbContext.js +70 -0
- package/src/integrations/performance.js +29 -3
- package/src/integrations/tracing.js +80 -0
- package/src/metricBuffer.js +117 -0
- package/src/performanceFlusher.js +43 -5
- package/src/reporter.js +3 -2
- package/src/spanBuffer.js +89 -0
|
@@ -0,0 +1,117 @@
|
|
|
1
|
+
/**
|
|
2
|
+
* Collects individual captureMetric()/captureInfrastructureMetric() calls in-process and periodically
|
|
3
|
+
* flushes them as one batch, rather than one network call per capture. Unlike performanceFlusher.js
|
|
4
|
+
* this keeps a *list* of individually meaningful entries instead of summing them into buckets: a
|
|
5
|
+
* customer's own signup or payment is exactly the kind of thing they will want a genuinely accurate
|
|
6
|
+
* count/sum of later, so the server stores one row per entry as-is. Ported from
|
|
7
|
+
* gems/forge_ops_tracker/lib/forge_ops_tracker/metric_buffer.rb and infrastructure_metric_buffer.rb,
|
|
8
|
+
* which are the same class twice; here it is one class instantiated twice, told which delivery
|
|
9
|
+
* method and flush interval to use.
|
|
10
|
+
*
|
|
11
|
+
* Three deliberate differences from the Ruby buffers:
|
|
12
|
+
*
|
|
13
|
+
* - A flush snapshots the first N entries and, on success, removes exactly those N, instead of
|
|
14
|
+
* resetting the whole list: an entry recorded while the request is in flight (this is async, so
|
|
15
|
+
* that window is real) is kept for the next flush rather than lost.
|
|
16
|
+
* - The buffer is capped at MAX_ENTRIES, and once full further entries are dropped until a flush
|
|
17
|
+
* succeeds: a plan without the feature answers 403 on every flush, and an uncapped buffer would
|
|
18
|
+
* then grow for as long as the process lives. Dropping the newest rather than the oldest keeps
|
|
19
|
+
* the entries a flush is delivering at the front of the list, which is what makes removing
|
|
20
|
+
* exactly those afterward exact.
|
|
21
|
+
* - A NaN or infinite value is dropped at record time: JSON.stringify turns it into `null`, which
|
|
22
|
+
* the server would reject, taking the whole batch with it.
|
|
23
|
+
*
|
|
24
|
+
* setInterval(...).unref(), not a real background thread, same reasoning sessionFlusher.js's own
|
|
25
|
+
* header comment documents. Unlike that flusher this also flushes on "beforeExit" (fired when the
|
|
26
|
+
* event loop has nothing left, and which does not change how the process exits the way a
|
|
27
|
+
* SIGTERM/SIGINT listener would): the real use of the infrastructure endpoint is a short-lived cron
|
|
28
|
+
* script that captures a few readings and lets the process end.
|
|
29
|
+
*/
|
|
30
|
+
export const MAX_ENTRIES = 1000;
|
|
31
|
+
|
|
32
|
+
export class MetricBuffer {
|
|
33
|
+
#configuration;
|
|
34
|
+
#deliver;
|
|
35
|
+
#intervalMs;
|
|
36
|
+
#entries = [];
|
|
37
|
+
#timer = null;
|
|
38
|
+
#beforeExit = null;
|
|
39
|
+
|
|
40
|
+
/**
|
|
41
|
+
* @param {import("./configuration.js").Configuration} configuration
|
|
42
|
+
* @param {(entries: Record<string, unknown>[]) => Promise<boolean>} deliver
|
|
43
|
+
* @param {() => number} intervalMs read fresh each time the timer is created
|
|
44
|
+
*/
|
|
45
|
+
constructor(configuration, deliver, intervalMs) {
|
|
46
|
+
this.#configuration = configuration;
|
|
47
|
+
this.#deliver = deliver;
|
|
48
|
+
this.#intervalMs = intervalMs;
|
|
49
|
+
}
|
|
50
|
+
|
|
51
|
+
/**
|
|
52
|
+
* @param {Record<string, unknown>} entry everything but recorded_at
|
|
53
|
+
* @returns {boolean} whether it was kept
|
|
54
|
+
*/
|
|
55
|
+
record(entry) {
|
|
56
|
+
const value = entry.value;
|
|
57
|
+
if (typeof value !== "number" || !Number.isFinite(value)) {
|
|
58
|
+
this.#configuration.log("[forge-ops-tracker] dropped a metric with a non-numeric or non-finite value");
|
|
59
|
+
return false;
|
|
60
|
+
}
|
|
61
|
+
if (this.#entries.length >= MAX_ENTRIES) {
|
|
62
|
+
this.#configuration.log("[forge-ops-tracker] metric buffer full, dropping a metric");
|
|
63
|
+
return false;
|
|
64
|
+
}
|
|
65
|
+
|
|
66
|
+
this.#ensureStarted();
|
|
67
|
+
this.#entries.push({ ...entry, recorded_at: `${new Date().toISOString().slice(0, 19)}Z` });
|
|
68
|
+
return true;
|
|
69
|
+
}
|
|
70
|
+
|
|
71
|
+
/** Delivers everything buffered so far as one batch. A failed delivery keeps every entry, so the next flush's batch just grows. */
|
|
72
|
+
async flush() {
|
|
73
|
+
if (this.#entries.length === 0) {
|
|
74
|
+
return;
|
|
75
|
+
}
|
|
76
|
+
|
|
77
|
+
const snapshot = this.#entries.slice();
|
|
78
|
+
const delivered = await this.#deliver(snapshot);
|
|
79
|
+
if (!delivered) {
|
|
80
|
+
return;
|
|
81
|
+
}
|
|
82
|
+
|
|
83
|
+
// Exactly the entries just delivered: anything recorded while the request was in flight sits
|
|
84
|
+
// after them and stays for the next flush.
|
|
85
|
+
this.#entries.splice(0, snapshot.length);
|
|
86
|
+
}
|
|
87
|
+
|
|
88
|
+
/** Stops the timer and the exit hook without delivering anything. */
|
|
89
|
+
discard() {
|
|
90
|
+
if (this.#timer !== null) {
|
|
91
|
+
clearInterval(this.#timer);
|
|
92
|
+
this.#timer = null;
|
|
93
|
+
}
|
|
94
|
+
if (this.#beforeExit !== null) {
|
|
95
|
+
process.off("beforeExit", this.#beforeExit);
|
|
96
|
+
this.#beforeExit = null;
|
|
97
|
+
}
|
|
98
|
+
this.#entries = [];
|
|
99
|
+
}
|
|
100
|
+
|
|
101
|
+
#ensureStarted() {
|
|
102
|
+
if (this.#timer !== null) {
|
|
103
|
+
return;
|
|
104
|
+
}
|
|
105
|
+
|
|
106
|
+
this.#timer = setInterval(() => this.#flushSafely(), this.#intervalMs());
|
|
107
|
+
this.#timer.unref();
|
|
108
|
+
this.#beforeExit = () => this.#flushSafely();
|
|
109
|
+
process.on("beforeExit", this.#beforeExit);
|
|
110
|
+
}
|
|
111
|
+
|
|
112
|
+
#flushSafely() {
|
|
113
|
+
return this.flush().catch((e) => {
|
|
114
|
+
this.#configuration.log(`[forge-ops-tracker] metric flush error: ${e.name}: ${e.message}`);
|
|
115
|
+
});
|
|
116
|
+
}
|
|
117
|
+
}
|
|
@@ -1,3 +1,5 @@
|
|
|
1
|
+
import { bucketFor } from "./histogramBucketer.js";
|
|
2
|
+
|
|
1
3
|
/**
|
|
2
4
|
* Times requests in-process, bucketed by transactionName (see integrations/performance.js), and
|
|
3
5
|
* periodically flushes each distinct bucket as one small aggregate report, rather than one
|
|
@@ -33,19 +35,30 @@ export class PerformanceFlusher {
|
|
|
33
35
|
record(transactionName, durationMs) {
|
|
34
36
|
this.#ensureTimerStarted();
|
|
35
37
|
|
|
36
|
-
const bucket = this.#buckets.get(transactionName) ?? { count: 0, durationSumMs: 0, maxDurationMs: 0 };
|
|
38
|
+
const bucket = this.#buckets.get(transactionName) ?? { count: 0, durationSumMs: 0, maxDurationMs: 0, histogram: {} };
|
|
37
39
|
bucket.count += 1;
|
|
38
40
|
bucket.durationSumMs += durationMs;
|
|
39
41
|
if (durationMs > bucket.maxDurationMs) {
|
|
40
42
|
bucket.maxDurationMs = durationMs;
|
|
41
43
|
}
|
|
44
|
+
// The distribution count/sum/max can't reconstruct: see histogramBucketer.js for why the
|
|
45
|
+
// server approximates a percentile from these bucket counts.
|
|
46
|
+
const label = bucketFor(durationMs);
|
|
47
|
+
bucket.histogram[label] = (bucket.histogram[label] ?? 0) + 1;
|
|
42
48
|
this.#buckets.set(transactionName, bucket);
|
|
43
49
|
}
|
|
44
50
|
|
|
45
51
|
/**
|
|
46
|
-
* Snapshots
|
|
47
|
-
*
|
|
48
|
-
*
|
|
52
|
+
* Snapshots the buckets, then delivers them as one batch. A failed delivery keeps every bucket
|
|
53
|
+
* where it is rather than resetting, so the next flush's batch just grows instead of losing what
|
|
54
|
+
* was already tallied; same reasoning sessionFlusher.js's own flush() documents.
|
|
55
|
+
*
|
|
56
|
+
* Only exactly what this snapshot delivered is removed afterward, subtracted from whatever is in
|
|
57
|
+
* each bucket by then, never the whole map reset: record() can run while the delivery is awaited,
|
|
58
|
+
* so a record for a transaction already in the snapshot, or a brand-new one, can land between the
|
|
59
|
+
* snapshot and delivery succeeding, and resetting afterward would silently discard it.
|
|
60
|
+
* maxDurationMs is left as whatever is currently on the bucket, sent or not: a max can't be
|
|
61
|
+
* "subtracted" back out, and leaving it never overstates the next period's own max.
|
|
49
62
|
*/
|
|
50
63
|
async flush() {
|
|
51
64
|
if (this.#buckets.size === 0) {
|
|
@@ -53,6 +66,12 @@ export class PerformanceFlusher {
|
|
|
53
66
|
}
|
|
54
67
|
|
|
55
68
|
const periodEndedAt = new Date();
|
|
69
|
+
const sent = new Map(
|
|
70
|
+
[...this.#buckets.entries()].map(([transactionName, bucket]) => [
|
|
71
|
+
transactionName,
|
|
72
|
+
{ count: bucket.count, durationSumMs: bucket.durationSumMs, histogram: { ...bucket.histogram } },
|
|
73
|
+
]),
|
|
74
|
+
);
|
|
56
75
|
const samples = [...this.#buckets.entries()].map(([transactionName, bucket]) => ({
|
|
57
76
|
transaction_name: transactionName,
|
|
58
77
|
environment: this.#configuration.environment,
|
|
@@ -62,6 +81,7 @@ export class PerformanceFlusher {
|
|
|
62
81
|
request_count: bucket.count,
|
|
63
82
|
duration_sum_ms: bucket.durationSumMs,
|
|
64
83
|
max_duration_ms: bucket.maxDurationMs,
|
|
84
|
+
histogram: { ...bucket.histogram },
|
|
65
85
|
}));
|
|
66
86
|
|
|
67
87
|
const delivered = await this.#client.deliverPerformanceSamples(samples);
|
|
@@ -69,7 +89,25 @@ export class PerformanceFlusher {
|
|
|
69
89
|
return;
|
|
70
90
|
}
|
|
71
91
|
|
|
72
|
-
|
|
92
|
+
for (const [transactionName, deliveredBucket] of sent) {
|
|
93
|
+
const current = this.#buckets.get(transactionName);
|
|
94
|
+
if (!current) {
|
|
95
|
+
continue;
|
|
96
|
+
}
|
|
97
|
+
current.count -= deliveredBucket.count;
|
|
98
|
+
current.durationSumMs = Math.max(current.durationSumMs - deliveredBucket.durationSumMs, 0);
|
|
99
|
+
for (const [label, count] of Object.entries(deliveredBucket.histogram)) {
|
|
100
|
+
const remaining = (current.histogram[label] ?? 0) - count;
|
|
101
|
+
if (remaining > 0) {
|
|
102
|
+
current.histogram[label] = remaining;
|
|
103
|
+
} else {
|
|
104
|
+
delete current.histogram[label];
|
|
105
|
+
}
|
|
106
|
+
}
|
|
107
|
+
if (current.count <= 0) {
|
|
108
|
+
this.#buckets.delete(transactionName);
|
|
109
|
+
}
|
|
110
|
+
}
|
|
73
111
|
this.#periodStartedAt = periodEndedAt;
|
|
74
112
|
}
|
|
75
113
|
|
package/src/reporter.js
CHANGED
|
@@ -21,14 +21,15 @@ export class Reporter {
|
|
|
21
21
|
* @param {Error} error
|
|
22
22
|
* @param {Record<string, unknown>} [context]
|
|
23
23
|
* @param {Record<string, unknown> | null} [user]
|
|
24
|
+
* @param {Array<Record<string, unknown>>} [breadcrumbs]
|
|
24
25
|
*/
|
|
25
|
-
report(error, context = {}, user = null) {
|
|
26
|
+
report(error, context = {}, user = null, breadcrumbs = []) {
|
|
26
27
|
try {
|
|
27
28
|
if (!this.#configuration.isEnabled()) {
|
|
28
29
|
return;
|
|
29
30
|
}
|
|
30
31
|
|
|
31
|
-
const payload = this.#eventBuilder.build(error, context, user);
|
|
32
|
+
const payload = this.#eventBuilder.build(error, context, user, breadcrumbs);
|
|
32
33
|
this.#deliveryQueue.push(payload);
|
|
33
34
|
} catch (e) {
|
|
34
35
|
this.#configuration.log(`[forge-ops-tracker] report failed: ${e.name}: ${e.message}`);
|
|
@@ -0,0 +1,89 @@
|
|
|
1
|
+
import { randomBytes } from "node:crypto";
|
|
2
|
+
|
|
3
|
+
/**
|
|
4
|
+
* A short, unique-enough identifier for one span: 8 random bytes as hex, matching
|
|
5
|
+
* gems/forge_ops_tracker/lib/forge_ops_tracker/span_buffer.rb's own SecureRandom.hex(8). The
|
|
6
|
+
* server only ever needs these to be unique within one trace's own array (see
|
|
7
|
+
* Api::V1::SpansController on the Rails side), never a real database id, so this is deliberately
|
|
8
|
+
* cheap rather than a full UUID.
|
|
9
|
+
* @returns {string}
|
|
10
|
+
*/
|
|
11
|
+
export function randomSpanId() {
|
|
12
|
+
return randomBytes(8).toString("hex");
|
|
13
|
+
}
|
|
14
|
+
|
|
15
|
+
/**
|
|
16
|
+
* Accumulates one request's own nested call tree in-process: the tracing analog to
|
|
17
|
+
* BreadcrumbBuffer's per-request trail. Ported from
|
|
18
|
+
* gems/forge_ops_tracker/lib/forge_ops_tracker/span_buffer.rb, with one deliberate departure: the
|
|
19
|
+
* Ruby buffer tracks "what's currently open" with a single mutable stack, which is correct for
|
|
20
|
+
* Ruby's own single-threaded-per-request execution model, but is the wrong tool here. A customer
|
|
21
|
+
* can easily await two service-layer span() calls concurrently in the same request (Promise.all
|
|
22
|
+
* rather than one after another); popping one call's own span off a shared stack while the other
|
|
23
|
+
* is still pending would leave the second one mis-parented under the first instead of under the
|
|
24
|
+
* request's own root. index.js's own span()/leaf-span recording instead tracks "the currently
|
|
25
|
+
* open span" with a second, nested AsyncLocalStorage (see spanParentStorage there): each
|
|
26
|
+
* concurrent branch gets its own correctly-scoped view automatically, the exact property that
|
|
27
|
+
* storage exists to provide. This class only ever holds what doesn't depend on that: the trace's
|
|
28
|
+
* own identifiers and the flat list every span, root or leaf, ends up recorded into.
|
|
29
|
+
*/
|
|
30
|
+
export class SpanBuffer {
|
|
31
|
+
#configuration;
|
|
32
|
+
traceId;
|
|
33
|
+
rootSpanId;
|
|
34
|
+
spans = [];
|
|
35
|
+
#rootDurationMs = null;
|
|
36
|
+
|
|
37
|
+
constructor(configuration) {
|
|
38
|
+
this.#configuration = configuration;
|
|
39
|
+
this.traceId = randomBytes(16).toString("hex");
|
|
40
|
+
this.rootSpanId = randomSpanId();
|
|
41
|
+
}
|
|
42
|
+
|
|
43
|
+
/**
|
|
44
|
+
* root: true only for the one call that records the request's own root span: forces
|
|
45
|
+
* parent_span_id null explicitly, the same reasoning span_buffer.rb's own #record documents for
|
|
46
|
+
* why that can't just be read off of "whatever's currently open" for the root specifically.
|
|
47
|
+
*
|
|
48
|
+
* @param {{
|
|
49
|
+
* spanId: string,
|
|
50
|
+
* parentSpanId?: string | null,
|
|
51
|
+
* name: string,
|
|
52
|
+
* kind: string,
|
|
53
|
+
* startedAt: Date,
|
|
54
|
+
* durationMs: number,
|
|
55
|
+
* data?: Record<string, unknown>,
|
|
56
|
+
* root?: boolean,
|
|
57
|
+
* }} span
|
|
58
|
+
*/
|
|
59
|
+
record({ spanId, parentSpanId = null, name, kind, startedAt, durationMs, data = {}, root = false }) {
|
|
60
|
+
this.spans.push({
|
|
61
|
+
span_id: spanId,
|
|
62
|
+
parent_span_id: root ? null : parentSpanId,
|
|
63
|
+
name,
|
|
64
|
+
kind,
|
|
65
|
+
// Date#toISOString() already yields millisecond precision UTC ("...sssZ"), exactly the wire
|
|
66
|
+
// format the server expects; unlike BreadcrumbBuffer's own timestamp, this one must NOT
|
|
67
|
+
// strip milliseconds, since a waterfall's own ordering depends on them.
|
|
68
|
+
started_at: startedAt.toISOString(),
|
|
69
|
+
duration_ms: durationMs,
|
|
70
|
+
environment: this.#configuration.environment,
|
|
71
|
+
release: this.#configuration.release,
|
|
72
|
+
data,
|
|
73
|
+
});
|
|
74
|
+
if (root) {
|
|
75
|
+
this.#rootDurationMs = durationMs;
|
|
76
|
+
}
|
|
77
|
+
}
|
|
78
|
+
|
|
79
|
+
/**
|
|
80
|
+
* null (never sent) until the root span has actually been recorded: a request whose own
|
|
81
|
+
* tracing integration never got the chance to call back at all has no duration to compare
|
|
82
|
+
* against a threshold, so it's never mistakenly treated as slow.
|
|
83
|
+
* @param {number} thresholdMs
|
|
84
|
+
* @returns {boolean}
|
|
85
|
+
*/
|
|
86
|
+
isSlow(thresholdMs) {
|
|
87
|
+
return this.#rootDurationMs !== null && this.#rootDurationMs >= thresholdMs;
|
|
88
|
+
}
|
|
89
|
+
}
|