@forge-ops/tracker 0.3.0 → 0.9.0
This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
- package/README.md +201 -8
- package/package.json +4 -2
- package/src/breadcrumbBuffer.js +48 -0
- package/src/client.js +29 -0
- package/src/configuration.js +73 -0
- package/src/deliveryQueue.js +16 -3
- package/src/eventBuilder.js +24 -5
- package/src/histogramBucketer.js +26 -0
- package/src/httpTracing.js +104 -0
- package/src/index.js +409 -3
- package/src/integrations/breadcrumbContext.js +70 -0
- package/src/integrations/express.js +50 -1
- package/src/integrations/performance.js +29 -3
- package/src/integrations/tracing.js +80 -0
- package/src/metricBuffer.js +117 -0
- package/src/performanceFlusher.js +43 -5
- package/src/reporter.js +4 -2
- package/src/spanBuffer.js +89 -0
|
@@ -1,4 +1,4 @@
|
|
|
1
|
-
import { captureException } from "../index.js";
|
|
1
|
+
import { captureException, runWithUser } from "../index.js";
|
|
2
2
|
|
|
3
3
|
/**
|
|
4
4
|
* Express error-handling middleware. Register last, after all routes:
|
|
@@ -26,3 +26,52 @@ export function forgeOpsTrackerExpressMiddleware(err, req, res, next) {
|
|
|
26
26
|
});
|
|
27
27
|
next(err);
|
|
28
28
|
}
|
|
29
|
+
|
|
30
|
+
/**
|
|
31
|
+
* Automatically identifies the affected user for every error reported for the rest of this
|
|
32
|
+
* request, when `req.user` is present. Register early, before your routes (and after whatever
|
|
33
|
+
* middleware actually sets `req.user`, e.g. Passport's own session middleware):
|
|
34
|
+
*
|
|
35
|
+
* app.use(passport.session());
|
|
36
|
+
* app.use(forgeOpsTrackerUserContextMiddleware);
|
|
37
|
+
* // ...routes...
|
|
38
|
+
* app.use(forgeOpsTrackerExpressMiddleware);
|
|
39
|
+
*
|
|
40
|
+
* `req.user` is Passport's own convention (the closest thing Express has to a single dominant
|
|
41
|
+
* auth library, the same role Warden plays for Rails): set once `passport.authenticate()`/session
|
|
42
|
+
* deserialization succeeds, left `undefined` otherwise, so a plain truthy check is enough, the
|
|
43
|
+
* same duck-typed "is there a user object at all" check `gems/forge_ops_tracker`'s own Warden
|
|
44
|
+
* integration does for `env["warden"].user`. A no-op for an app that never set `req.user` at all,
|
|
45
|
+
* whether that's because nobody's signed in or Passport (or an equivalent) isn't installed.
|
|
46
|
+
*
|
|
47
|
+
* Wraps the rest of the request in `runWithUser()` (see that function's own doc comment for why
|
|
48
|
+
* this SDK has no imperative `setUser()`) rather than setting anything imperatively itself:
|
|
49
|
+
* confirmed directly, with a real Express app and a real async route handler, that
|
|
50
|
+
* `AsyncLocalStorage`'s context set by an early middleware like this one does survive all the way
|
|
51
|
+
* through to a later error-handling middleware (`forgeOpsTrackerExpressMiddleware`, typically
|
|
52
|
+
* registered last, after every route) in the same request, not just to handlers registered
|
|
53
|
+
* synchronously right after this one.
|
|
54
|
+
*/
|
|
55
|
+
export function forgeOpsTrackerUserContextMiddleware(req, res, next) {
|
|
56
|
+
if (!req.user) {
|
|
57
|
+
next();
|
|
58
|
+
return;
|
|
59
|
+
}
|
|
60
|
+
runWithUser(serializeUser(req.user), next);
|
|
61
|
+
}
|
|
62
|
+
|
|
63
|
+
function serializeUser(user) {
|
|
64
|
+
const result = {};
|
|
65
|
+
if (user.id !== undefined && user.id !== null) {
|
|
66
|
+
result.id = String(user.id);
|
|
67
|
+
}
|
|
68
|
+
if (user.email) {
|
|
69
|
+
result.email = user.email;
|
|
70
|
+
}
|
|
71
|
+
if (user.username) {
|
|
72
|
+
result.username = user.username;
|
|
73
|
+
} else if (user.name) {
|
|
74
|
+
result.username = user.name;
|
|
75
|
+
}
|
|
76
|
+
return result;
|
|
77
|
+
}
|
|
@@ -1,10 +1,11 @@
|
|
|
1
|
-
import { _recordPerformance } from "../index.js";
|
|
1
|
+
import { _recordBreadcrumb, _recordPerformance } from "../index.js";
|
|
2
2
|
|
|
3
3
|
/**
|
|
4
4
|
* Express performance-tracking middleware. Register wherever is convenient relative to routes
|
|
5
5
|
* (unlike sessionTracking.js's own middleware, this doesn't need to run first: it only reads
|
|
6
6
|
* req.route once the request has already been routed, inside the "finish" listener below, not
|
|
7
|
-
* at request-start)
|
|
7
|
+
* at request-start), but after forgeOpsTrackerBreadcrumbContextExpressMiddleware (see
|
|
8
|
+
* integrations/breadcrumbContext.js), so the breadcrumb recorded below has a trail to land in:
|
|
8
9
|
*
|
|
9
10
|
* app.use(forgeOpsTrackerPerformanceExpressMiddleware);
|
|
10
11
|
*
|
|
@@ -17,9 +18,30 @@ import { _recordPerformance } from "../index.js";
|
|
|
17
18
|
* transactionName is "<HTTP method> <route pattern>", e.g. "GET /users/:id", not the raw URL:
|
|
18
19
|
* req.route.path is Express's own matched route pattern, read here (after routing has actually
|
|
19
20
|
* happened, unlike at request-start where it isn't populated yet), which keeps a distinct user
|
|
20
|
-
* id from exploding into its own separate transaction the way the literal URL would
|
|
21
|
+
* id from exploding into its own separate transaction the way the literal URL would: the
|
|
21
22
|
* low-cardinality equivalent of a Rails controller#action. Falls back to req.path (the literal,
|
|
22
23
|
* unmatched path) for a request that never matched a route at all (a 404).
|
|
24
|
+
*
|
|
25
|
+
* The same res.on("finish") listener also records a "controller" breadcrumb, gated on
|
|
26
|
+
* trackBreadcrumbs independently of trackPerformance (each call below checks its own flag,
|
|
27
|
+
* mirroring gems/forge_ops_tracker/lib/forge_ops_tracker/railtie.rb's own
|
|
28
|
+
* process_action.action_controller subscription, which records both a performance sample and a
|
|
29
|
+
* breadcrumb from the one event): an app could want the trail without the timing data, or vice
|
|
30
|
+
* versa. This is currently the only place this client automatically records a breadcrumb at all:
|
|
31
|
+
* unlike Rails' framework-wide ActiveSupport::Notifications, this client has no automatic
|
|
32
|
+
* instrumentation source that isn't also this same opt-in middleware, so an app that wants
|
|
33
|
+
* automatic breadcrumbs (not just manual ones from addBreadcrumb()) needs this middleware
|
|
34
|
+
* installed regardless of whether it also wants the performance data. There's no query-level
|
|
35
|
+
* automatic breadcrumb source for the same reason there's no query-level performance
|
|
36
|
+
* instrumentation in this client at all today (no ORM/DB driver integration exists yet to hang
|
|
37
|
+
* one off of); this doesn't invent one.
|
|
38
|
+
*
|
|
39
|
+
* Deliberately recorded with the request's final status already known (from the same
|
|
40
|
+
* res.on("finish") firing this middleware already needed for timing), the same "recorded once
|
|
41
|
+
* the response actually finished" shape as Ruby's/Python's own automatic controller breadcrumb;
|
|
42
|
+
* see those implementations' own documented gap (a breadcrumb recorded this late can never show
|
|
43
|
+
* up on that exact same request's own error, only a later one on the same connection/process
|
|
44
|
+
* could see it) for why that's an accepted trade-off, not an oversight here either.
|
|
23
45
|
*/
|
|
24
46
|
export function forgeOpsTrackerPerformanceExpressMiddleware(req, res, next) {
|
|
25
47
|
const start = performance.now();
|
|
@@ -28,6 +50,10 @@ export function forgeOpsTrackerPerformanceExpressMiddleware(req, res, next) {
|
|
|
28
50
|
const durationMs = performance.now() - start;
|
|
29
51
|
const transactionName = `${req.method} ${req.route?.path ?? req.path}`;
|
|
30
52
|
_recordPerformance(transactionName, durationMs);
|
|
53
|
+
_recordBreadcrumb(transactionName, "controller", res.statusCode >= 500 ? "error" : "info", {
|
|
54
|
+
status: res.statusCode,
|
|
55
|
+
duration_ms: Math.round(durationMs * 10) / 10,
|
|
56
|
+
});
|
|
31
57
|
});
|
|
32
58
|
|
|
33
59
|
next();
|
|
@@ -0,0 +1,80 @@
|
|
|
1
|
+
import { _finishSpanTrace, _runWithSpanTrace } from "../index.js";
|
|
2
|
+
|
|
3
|
+
/**
|
|
4
|
+
* Express distributed-tracing middleware: wraps the rest of the request in _runWithSpanTrace
|
|
5
|
+
* (see that function's own comment in index.js), the same "own file, own registration point"
|
|
6
|
+
* shape breadcrumbContext.js already established for a genuinely distinct concern (this is
|
|
7
|
+
* separate from, and independent of, error capture, session tracking, and aggregate performance
|
|
8
|
+
* timing). Register wherever is convenient relative to routes, the same as
|
|
9
|
+
* forgeOpsTrackerPerformanceExpressMiddleware, since it only reads req.route once the request has
|
|
10
|
+
* already been routed, inside the "finish" listener below, not at request-start:
|
|
11
|
+
*
|
|
12
|
+
* app.use(forgeOpsTrackerTracingExpressMiddleware);
|
|
13
|
+
*
|
|
14
|
+
* transactionName is built exactly like forgeOpsTrackerPerformanceExpressMiddleware's own
|
|
15
|
+
* ("<HTTP method> <route pattern>", never the raw URL): this is the same request a performance
|
|
16
|
+
* sample would already be recorded for, so a trace's own root span name has to match, not invent
|
|
17
|
+
* a second naming scheme for the same thing.
|
|
18
|
+
*
|
|
19
|
+
* The actual send-or-don't decision happens inside _finishSpanTrace, only once the response has
|
|
20
|
+
* actually finished and the root span's real total duration is known: nothing about a normal,
|
|
21
|
+
* fast request costs a single byte over the wire, matching gems/forge_ops_tracker's own
|
|
22
|
+
* Middleware::SpanTracing.
|
|
23
|
+
*/
|
|
24
|
+
export function forgeOpsTrackerTracingExpressMiddleware(req, res, next) {
|
|
25
|
+
_runWithSpanTrace(() => {
|
|
26
|
+
const startedAt = new Date();
|
|
27
|
+
const start = performance.now();
|
|
28
|
+
|
|
29
|
+
res.on("finish", () => {
|
|
30
|
+
const durationMs = performance.now() - start;
|
|
31
|
+
const transactionName = `${req.method} ${req.route?.path ?? req.path}`;
|
|
32
|
+
_finishSpanTrace(transactionName, startedAt, durationMs);
|
|
33
|
+
});
|
|
34
|
+
|
|
35
|
+
next();
|
|
36
|
+
});
|
|
37
|
+
}
|
|
38
|
+
|
|
39
|
+
/**
|
|
40
|
+
* Fastify equivalent, added as two separate hooks rather than one function wrapping "the rest of
|
|
41
|
+
* this request" the way Express's next() allows (see
|
|
42
|
+
* registerForgeOpsTrackerBreadcrumbContext's own comment in breadcrumbContext.js for why Fastify
|
|
43
|
+
* needs this shape at all): onRequest establishes the span trace (wrapping `done`, the same
|
|
44
|
+
* pattern that file already uses) and stashes this request's own start time on `request` itself,
|
|
45
|
+
* the same "flag stashed on the request object" convention integrations/express.js's own
|
|
46
|
+
* req._forgeOpsSessionCrashed already established; onResponse, Fastify's own dedicated
|
|
47
|
+
* "the response has actually been sent" hook, reads that back once the real total duration is
|
|
48
|
+
* known:
|
|
49
|
+
*
|
|
50
|
+
* registerForgeOpsTrackerTracing(app); // alongside registerForgeOpsTrackerBreadcrumbContext
|
|
51
|
+
*
|
|
52
|
+
* transactionName reads request.routeOptions.url, Fastify's own matched route pattern (the
|
|
53
|
+
* property that replaced the older, now-removed request.routerPath), falling back to routerPath
|
|
54
|
+
* for an older Fastify version and finally to the literal request.url for a request that never
|
|
55
|
+
* matched a route at all (a 404), the same "pattern first, literal path as the 404 fallback"
|
|
56
|
+
* shape the Express integration above already takes.
|
|
57
|
+
*
|
|
58
|
+
* @param {import("fastify").FastifyInstance} fastify
|
|
59
|
+
*/
|
|
60
|
+
export function registerForgeOpsTrackerTracing(fastify) {
|
|
61
|
+
fastify.addHook("onRequest", (request, reply, done) => {
|
|
62
|
+
_runWithSpanTrace(() => {
|
|
63
|
+
request._forgeOpsSpanStartedAt = new Date();
|
|
64
|
+
request._forgeOpsSpanStart = performance.now();
|
|
65
|
+
done();
|
|
66
|
+
});
|
|
67
|
+
});
|
|
68
|
+
|
|
69
|
+
fastify.addHook("onResponse", async (request) => {
|
|
70
|
+
if (request._forgeOpsSpanStart === undefined) {
|
|
71
|
+
// trackTracing was off (or the client wasn't enabled) when onRequest ran above, so there's
|
|
72
|
+
// nothing to finish; _finishSpanTrace would no-op anyway (no buffer to read back), but
|
|
73
|
+
// there's no point computing a duration for nothing.
|
|
74
|
+
return;
|
|
75
|
+
}
|
|
76
|
+
const durationMs = performance.now() - request._forgeOpsSpanStart;
|
|
77
|
+
const transactionName = `${request.method} ${request.routeOptions?.url ?? request.routerPath ?? request.url}`;
|
|
78
|
+
_finishSpanTrace(transactionName, request._forgeOpsSpanStartedAt, durationMs);
|
|
79
|
+
});
|
|
80
|
+
}
|
|
@@ -0,0 +1,117 @@
|
|
|
1
|
+
/**
|
|
2
|
+
* Collects individual captureMetric()/captureInfrastructureMetric() calls in-process and periodically
|
|
3
|
+
* flushes them as one batch, rather than one network call per capture. Unlike performanceFlusher.js
|
|
4
|
+
* this keeps a *list* of individually meaningful entries instead of summing them into buckets: a
|
|
5
|
+
* customer's own signup or payment is exactly the kind of thing they will want a genuinely accurate
|
|
6
|
+
* count/sum of later, so the server stores one row per entry as-is. Ported from
|
|
7
|
+
* gems/forge_ops_tracker/lib/forge_ops_tracker/metric_buffer.rb and infrastructure_metric_buffer.rb,
|
|
8
|
+
* which are the same class twice; here it is one class instantiated twice, told which delivery
|
|
9
|
+
* method and flush interval to use.
|
|
10
|
+
*
|
|
11
|
+
* Three deliberate differences from the Ruby buffers:
|
|
12
|
+
*
|
|
13
|
+
* - A flush snapshots the first N entries and, on success, removes exactly those N, instead of
|
|
14
|
+
* resetting the whole list: an entry recorded while the request is in flight (this is async, so
|
|
15
|
+
* that window is real) is kept for the next flush rather than lost.
|
|
16
|
+
* - The buffer is capped at MAX_ENTRIES, and once full further entries are dropped until a flush
|
|
17
|
+
* succeeds: a plan without the feature answers 403 on every flush, and an uncapped buffer would
|
|
18
|
+
* then grow for as long as the process lives. Dropping the newest rather than the oldest keeps
|
|
19
|
+
* the entries a flush is delivering at the front of the list, which is what makes removing
|
|
20
|
+
* exactly those afterward exact.
|
|
21
|
+
* - A NaN or infinite value is dropped at record time: JSON.stringify turns it into `null`, which
|
|
22
|
+
* the server would reject, taking the whole batch with it.
|
|
23
|
+
*
|
|
24
|
+
* setInterval(...).unref(), not a real background thread, same reasoning sessionFlusher.js's own
|
|
25
|
+
* header comment documents. Unlike that flusher this also flushes on "beforeExit" (fired when the
|
|
26
|
+
* event loop has nothing left, and which does not change how the process exits the way a
|
|
27
|
+
* SIGTERM/SIGINT listener would): the real use of the infrastructure endpoint is a short-lived cron
|
|
28
|
+
* script that captures a few readings and lets the process end.
|
|
29
|
+
*/
|
|
30
|
+
export const MAX_ENTRIES = 1000;
|
|
31
|
+
|
|
32
|
+
export class MetricBuffer {
|
|
33
|
+
#configuration;
|
|
34
|
+
#deliver;
|
|
35
|
+
#intervalMs;
|
|
36
|
+
#entries = [];
|
|
37
|
+
#timer = null;
|
|
38
|
+
#beforeExit = null;
|
|
39
|
+
|
|
40
|
+
/**
|
|
41
|
+
* @param {import("./configuration.js").Configuration} configuration
|
|
42
|
+
* @param {(entries: Record<string, unknown>[]) => Promise<boolean>} deliver
|
|
43
|
+
* @param {() => number} intervalMs read fresh each time the timer is created
|
|
44
|
+
*/
|
|
45
|
+
constructor(configuration, deliver, intervalMs) {
|
|
46
|
+
this.#configuration = configuration;
|
|
47
|
+
this.#deliver = deliver;
|
|
48
|
+
this.#intervalMs = intervalMs;
|
|
49
|
+
}
|
|
50
|
+
|
|
51
|
+
/**
|
|
52
|
+
* @param {Record<string, unknown>} entry everything but recorded_at
|
|
53
|
+
* @returns {boolean} whether it was kept
|
|
54
|
+
*/
|
|
55
|
+
record(entry) {
|
|
56
|
+
const value = entry.value;
|
|
57
|
+
if (typeof value !== "number" || !Number.isFinite(value)) {
|
|
58
|
+
this.#configuration.log("[forge-ops-tracker] dropped a metric with a non-numeric or non-finite value");
|
|
59
|
+
return false;
|
|
60
|
+
}
|
|
61
|
+
if (this.#entries.length >= MAX_ENTRIES) {
|
|
62
|
+
this.#configuration.log("[forge-ops-tracker] metric buffer full, dropping a metric");
|
|
63
|
+
return false;
|
|
64
|
+
}
|
|
65
|
+
|
|
66
|
+
this.#ensureStarted();
|
|
67
|
+
this.#entries.push({ ...entry, recorded_at: `${new Date().toISOString().slice(0, 19)}Z` });
|
|
68
|
+
return true;
|
|
69
|
+
}
|
|
70
|
+
|
|
71
|
+
/** Delivers everything buffered so far as one batch. A failed delivery keeps every entry, so the next flush's batch just grows. */
|
|
72
|
+
async flush() {
|
|
73
|
+
if (this.#entries.length === 0) {
|
|
74
|
+
return;
|
|
75
|
+
}
|
|
76
|
+
|
|
77
|
+
const snapshot = this.#entries.slice();
|
|
78
|
+
const delivered = await this.#deliver(snapshot);
|
|
79
|
+
if (!delivered) {
|
|
80
|
+
return;
|
|
81
|
+
}
|
|
82
|
+
|
|
83
|
+
// Exactly the entries just delivered: anything recorded while the request was in flight sits
|
|
84
|
+
// after them and stays for the next flush.
|
|
85
|
+
this.#entries.splice(0, snapshot.length);
|
|
86
|
+
}
|
|
87
|
+
|
|
88
|
+
/** Stops the timer and the exit hook without delivering anything. */
|
|
89
|
+
discard() {
|
|
90
|
+
if (this.#timer !== null) {
|
|
91
|
+
clearInterval(this.#timer);
|
|
92
|
+
this.#timer = null;
|
|
93
|
+
}
|
|
94
|
+
if (this.#beforeExit !== null) {
|
|
95
|
+
process.off("beforeExit", this.#beforeExit);
|
|
96
|
+
this.#beforeExit = null;
|
|
97
|
+
}
|
|
98
|
+
this.#entries = [];
|
|
99
|
+
}
|
|
100
|
+
|
|
101
|
+
#ensureStarted() {
|
|
102
|
+
if (this.#timer !== null) {
|
|
103
|
+
return;
|
|
104
|
+
}
|
|
105
|
+
|
|
106
|
+
this.#timer = setInterval(() => this.#flushSafely(), this.#intervalMs());
|
|
107
|
+
this.#timer.unref();
|
|
108
|
+
this.#beforeExit = () => this.#flushSafely();
|
|
109
|
+
process.on("beforeExit", this.#beforeExit);
|
|
110
|
+
}
|
|
111
|
+
|
|
112
|
+
#flushSafely() {
|
|
113
|
+
return this.flush().catch((e) => {
|
|
114
|
+
this.#configuration.log(`[forge-ops-tracker] metric flush error: ${e.name}: ${e.message}`);
|
|
115
|
+
});
|
|
116
|
+
}
|
|
117
|
+
}
|
|
@@ -1,3 +1,5 @@
|
|
|
1
|
+
import { bucketFor } from "./histogramBucketer.js";
|
|
2
|
+
|
|
1
3
|
/**
|
|
2
4
|
* Times requests in-process, bucketed by transactionName (see integrations/performance.js), and
|
|
3
5
|
* periodically flushes each distinct bucket as one small aggregate report, rather than one
|
|
@@ -33,19 +35,30 @@ export class PerformanceFlusher {
|
|
|
33
35
|
record(transactionName, durationMs) {
|
|
34
36
|
this.#ensureTimerStarted();
|
|
35
37
|
|
|
36
|
-
const bucket = this.#buckets.get(transactionName) ?? { count: 0, durationSumMs: 0, maxDurationMs: 0 };
|
|
38
|
+
const bucket = this.#buckets.get(transactionName) ?? { count: 0, durationSumMs: 0, maxDurationMs: 0, histogram: {} };
|
|
37
39
|
bucket.count += 1;
|
|
38
40
|
bucket.durationSumMs += durationMs;
|
|
39
41
|
if (durationMs > bucket.maxDurationMs) {
|
|
40
42
|
bucket.maxDurationMs = durationMs;
|
|
41
43
|
}
|
|
44
|
+
// The distribution count/sum/max can't reconstruct: see histogramBucketer.js for why the
|
|
45
|
+
// server approximates a percentile from these bucket counts.
|
|
46
|
+
const label = bucketFor(durationMs);
|
|
47
|
+
bucket.histogram[label] = (bucket.histogram[label] ?? 0) + 1;
|
|
42
48
|
this.#buckets.set(transactionName, bucket);
|
|
43
49
|
}
|
|
44
50
|
|
|
45
51
|
/**
|
|
46
|
-
* Snapshots
|
|
47
|
-
*
|
|
48
|
-
*
|
|
52
|
+
* Snapshots the buckets, then delivers them as one batch. A failed delivery keeps every bucket
|
|
53
|
+
* where it is rather than resetting, so the next flush's batch just grows instead of losing what
|
|
54
|
+
* was already tallied; same reasoning sessionFlusher.js's own flush() documents.
|
|
55
|
+
*
|
|
56
|
+
* Only exactly what this snapshot delivered is removed afterward, subtracted from whatever is in
|
|
57
|
+
* each bucket by then, never the whole map reset: record() can run while the delivery is awaited,
|
|
58
|
+
* so a record for a transaction already in the snapshot, or a brand-new one, can land between the
|
|
59
|
+
* snapshot and delivery succeeding, and resetting afterward would silently discard it.
|
|
60
|
+
* maxDurationMs is left as whatever is currently on the bucket, sent or not: a max can't be
|
|
61
|
+
* "subtracted" back out, and leaving it never overstates the next period's own max.
|
|
49
62
|
*/
|
|
50
63
|
async flush() {
|
|
51
64
|
if (this.#buckets.size === 0) {
|
|
@@ -53,6 +66,12 @@ export class PerformanceFlusher {
|
|
|
53
66
|
}
|
|
54
67
|
|
|
55
68
|
const periodEndedAt = new Date();
|
|
69
|
+
const sent = new Map(
|
|
70
|
+
[...this.#buckets.entries()].map(([transactionName, bucket]) => [
|
|
71
|
+
transactionName,
|
|
72
|
+
{ count: bucket.count, durationSumMs: bucket.durationSumMs, histogram: { ...bucket.histogram } },
|
|
73
|
+
]),
|
|
74
|
+
);
|
|
56
75
|
const samples = [...this.#buckets.entries()].map(([transactionName, bucket]) => ({
|
|
57
76
|
transaction_name: transactionName,
|
|
58
77
|
environment: this.#configuration.environment,
|
|
@@ -62,6 +81,7 @@ export class PerformanceFlusher {
|
|
|
62
81
|
request_count: bucket.count,
|
|
63
82
|
duration_sum_ms: bucket.durationSumMs,
|
|
64
83
|
max_duration_ms: bucket.maxDurationMs,
|
|
84
|
+
histogram: { ...bucket.histogram },
|
|
65
85
|
}));
|
|
66
86
|
|
|
67
87
|
const delivered = await this.#client.deliverPerformanceSamples(samples);
|
|
@@ -69,7 +89,25 @@ export class PerformanceFlusher {
|
|
|
69
89
|
return;
|
|
70
90
|
}
|
|
71
91
|
|
|
72
|
-
|
|
92
|
+
for (const [transactionName, deliveredBucket] of sent) {
|
|
93
|
+
const current = this.#buckets.get(transactionName);
|
|
94
|
+
if (!current) {
|
|
95
|
+
continue;
|
|
96
|
+
}
|
|
97
|
+
current.count -= deliveredBucket.count;
|
|
98
|
+
current.durationSumMs = Math.max(current.durationSumMs - deliveredBucket.durationSumMs, 0);
|
|
99
|
+
for (const [label, count] of Object.entries(deliveredBucket.histogram)) {
|
|
100
|
+
const remaining = (current.histogram[label] ?? 0) - count;
|
|
101
|
+
if (remaining > 0) {
|
|
102
|
+
current.histogram[label] = remaining;
|
|
103
|
+
} else {
|
|
104
|
+
delete current.histogram[label];
|
|
105
|
+
}
|
|
106
|
+
}
|
|
107
|
+
if (current.count <= 0) {
|
|
108
|
+
this.#buckets.delete(transactionName);
|
|
109
|
+
}
|
|
110
|
+
}
|
|
73
111
|
this.#periodStartedAt = periodEndedAt;
|
|
74
112
|
}
|
|
75
113
|
|
package/src/reporter.js
CHANGED
|
@@ -20,14 +20,16 @@ export class Reporter {
|
|
|
20
20
|
/**
|
|
21
21
|
* @param {Error} error
|
|
22
22
|
* @param {Record<string, unknown>} [context]
|
|
23
|
+
* @param {Record<string, unknown> | null} [user]
|
|
24
|
+
* @param {Array<Record<string, unknown>>} [breadcrumbs]
|
|
23
25
|
*/
|
|
24
|
-
report(error, context = {}) {
|
|
26
|
+
report(error, context = {}, user = null, breadcrumbs = []) {
|
|
25
27
|
try {
|
|
26
28
|
if (!this.#configuration.isEnabled()) {
|
|
27
29
|
return;
|
|
28
30
|
}
|
|
29
31
|
|
|
30
|
-
const payload = this.#eventBuilder.build(error, context);
|
|
32
|
+
const payload = this.#eventBuilder.build(error, context, user, breadcrumbs);
|
|
31
33
|
this.#deliveryQueue.push(payload);
|
|
32
34
|
} catch (e) {
|
|
33
35
|
this.#configuration.log(`[forge-ops-tracker] report failed: ${e.name}: ${e.message}`);
|
|
@@ -0,0 +1,89 @@
|
|
|
1
|
+
import { randomBytes } from "node:crypto";
|
|
2
|
+
|
|
3
|
+
/**
|
|
4
|
+
* A short, unique-enough identifier for one span: 8 random bytes as hex, matching
|
|
5
|
+
* gems/forge_ops_tracker/lib/forge_ops_tracker/span_buffer.rb's own SecureRandom.hex(8). The
|
|
6
|
+
* server only ever needs these to be unique within one trace's own array (see
|
|
7
|
+
* Api::V1::SpansController on the Rails side), never a real database id, so this is deliberately
|
|
8
|
+
* cheap rather than a full UUID.
|
|
9
|
+
* @returns {string}
|
|
10
|
+
*/
|
|
11
|
+
export function randomSpanId() {
|
|
12
|
+
return randomBytes(8).toString("hex");
|
|
13
|
+
}
|
|
14
|
+
|
|
15
|
+
/**
|
|
16
|
+
* Accumulates one request's own nested call tree in-process: the tracing analog to
|
|
17
|
+
* BreadcrumbBuffer's per-request trail. Ported from
|
|
18
|
+
* gems/forge_ops_tracker/lib/forge_ops_tracker/span_buffer.rb, with one deliberate departure: the
|
|
19
|
+
* Ruby buffer tracks "what's currently open" with a single mutable stack, which is correct for
|
|
20
|
+
* Ruby's own single-threaded-per-request execution model, but is the wrong tool here. A customer
|
|
21
|
+
* can easily await two service-layer span() calls concurrently in the same request (Promise.all
|
|
22
|
+
* rather than one after another); popping one call's own span off a shared stack while the other
|
|
23
|
+
* is still pending would leave the second one mis-parented under the first instead of under the
|
|
24
|
+
* request's own root. index.js's own span()/leaf-span recording instead tracks "the currently
|
|
25
|
+
* open span" with a second, nested AsyncLocalStorage (see spanParentStorage there): each
|
|
26
|
+
* concurrent branch gets its own correctly-scoped view automatically, the exact property that
|
|
27
|
+
* storage exists to provide. This class only ever holds what doesn't depend on that: the trace's
|
|
28
|
+
* own identifiers and the flat list every span, root or leaf, ends up recorded into.
|
|
29
|
+
*/
|
|
30
|
+
export class SpanBuffer {
|
|
31
|
+
#configuration;
|
|
32
|
+
traceId;
|
|
33
|
+
rootSpanId;
|
|
34
|
+
spans = [];
|
|
35
|
+
#rootDurationMs = null;
|
|
36
|
+
|
|
37
|
+
constructor(configuration) {
|
|
38
|
+
this.#configuration = configuration;
|
|
39
|
+
this.traceId = randomBytes(16).toString("hex");
|
|
40
|
+
this.rootSpanId = randomSpanId();
|
|
41
|
+
}
|
|
42
|
+
|
|
43
|
+
/**
|
|
44
|
+
* root: true only for the one call that records the request's own root span: forces
|
|
45
|
+
* parent_span_id null explicitly, the same reasoning span_buffer.rb's own #record documents for
|
|
46
|
+
* why that can't just be read off of "whatever's currently open" for the root specifically.
|
|
47
|
+
*
|
|
48
|
+
* @param {{
|
|
49
|
+
* spanId: string,
|
|
50
|
+
* parentSpanId?: string | null,
|
|
51
|
+
* name: string,
|
|
52
|
+
* kind: string,
|
|
53
|
+
* startedAt: Date,
|
|
54
|
+
* durationMs: number,
|
|
55
|
+
* data?: Record<string, unknown>,
|
|
56
|
+
* root?: boolean,
|
|
57
|
+
* }} span
|
|
58
|
+
*/
|
|
59
|
+
record({ spanId, parentSpanId = null, name, kind, startedAt, durationMs, data = {}, root = false }) {
|
|
60
|
+
this.spans.push({
|
|
61
|
+
span_id: spanId,
|
|
62
|
+
parent_span_id: root ? null : parentSpanId,
|
|
63
|
+
name,
|
|
64
|
+
kind,
|
|
65
|
+
// Date#toISOString() already yields millisecond precision UTC ("...sssZ"), exactly the wire
|
|
66
|
+
// format the server expects; unlike BreadcrumbBuffer's own timestamp, this one must NOT
|
|
67
|
+
// strip milliseconds, since a waterfall's own ordering depends on them.
|
|
68
|
+
started_at: startedAt.toISOString(),
|
|
69
|
+
duration_ms: durationMs,
|
|
70
|
+
environment: this.#configuration.environment,
|
|
71
|
+
release: this.#configuration.release,
|
|
72
|
+
data,
|
|
73
|
+
});
|
|
74
|
+
if (root) {
|
|
75
|
+
this.#rootDurationMs = durationMs;
|
|
76
|
+
}
|
|
77
|
+
}
|
|
78
|
+
|
|
79
|
+
/**
|
|
80
|
+
* null (never sent) until the root span has actually been recorded: a request whose own
|
|
81
|
+
* tracing integration never got the chance to call back at all has no duration to compare
|
|
82
|
+
* against a threshold, so it's never mistakenly treated as slow.
|
|
83
|
+
* @param {number} thresholdMs
|
|
84
|
+
* @returns {boolean}
|
|
85
|
+
*/
|
|
86
|
+
isSlow(thresholdMs) {
|
|
87
|
+
return this.#rootDurationMs !== null && this.#rootDurationMs >= thresholdMs;
|
|
88
|
+
}
|
|
89
|
+
}
|