@ciphyrshq/sdk 2.6.0 → 3.0.0
This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
- package/README.md +131 -0
- package/package.json +4 -2
- package/src/client.js +326 -20
- package/src/client.test.js +99 -0
- package/src/context.js +82 -0
- package/src/fail-posture.js +100 -0
- package/src/fail-posture.test.js +385 -0
- package/src/index.js +14 -0
- package/src/no-network.test-helper.js +108 -0
- package/src/propagation.js +388 -0
- package/src/propagation.test.js +429 -0
- package/src/protect-tool.js +362 -0
- package/src/protect-tool.test.js +446 -0
- package/src/secret-detector.js +21 -3
- package/src/secret-detector.test.js +155 -0
- package/src/tracer.js +278 -9
- package/src/tracer.test.js +194 -0
- package/types.d.ts +265 -5
package/src/tracer.js
CHANGED
|
@@ -1,4 +1,15 @@
|
|
|
1
|
-
import {
|
|
1
|
+
import { randomBytes } from 'node:crypto';
|
|
2
|
+
import { availableParallelism } from 'node:os';
|
|
3
|
+
import { spanStorage, traceStorage } from './context.js';
|
|
4
|
+
import { extract, registerInternalOrigin, setDefaultProject, autoInstrument } from './propagation.js';
|
|
5
|
+
import { remoteParent } from './context.js';
|
|
6
|
+
|
|
7
|
+
// W3C-shaped ids (32 / 16 hex). `traceparent` accepts nothing else, and
|
|
8
|
+
// generating them in that shape is what lets the header carry the REAL ids
|
|
9
|
+
// instead of a hash of them — so an OpenTelemetry-instrumented peer links to
|
|
10
|
+
// this exact trace. (randomUUID's dashes made it unusable as a W3C id.)
|
|
11
|
+
const newTraceId = () => randomBytes(16).toString('hex');
|
|
12
|
+
const newSpanId = () => randomBytes(8).toString('hex');
|
|
2
13
|
|
|
3
14
|
// ─────────────────────────────────────────────────────────────────────────────
|
|
4
15
|
// CiphyrsTracer — buffered span/metric collection with auto-flush
|
|
@@ -12,6 +23,14 @@ export class CiphyrsTracer {
|
|
|
12
23
|
#spanBuffer = [];
|
|
13
24
|
#metricBuffer = [];
|
|
14
25
|
#flushInterval;
|
|
26
|
+
#heartbeatInterval;
|
|
27
|
+
#knownAgents = new Set();
|
|
28
|
+
#spansSinceBeat = 0;
|
|
29
|
+
#errorsSinceBeat = 0;
|
|
30
|
+
#startedAt = Date.now();
|
|
31
|
+
#lastCpu = null;
|
|
32
|
+
#lastCpuAt = null;
|
|
33
|
+
#lastLag = null;
|
|
15
34
|
|
|
16
35
|
/**
|
|
17
36
|
* @param {import('./client.js').CiphyrsClient} client
|
|
@@ -19,29 +38,160 @@ export class CiphyrsTracer {
|
|
|
19
38
|
* @param {string} opts.projectName — Project name for grouping traces
|
|
20
39
|
* @param {number} [opts.flushIntervalMs=5000] — Auto-flush interval in ms
|
|
21
40
|
* @param {number} [opts.batchSize=50] — Flush when buffer reaches this size
|
|
41
|
+
* @param {string} [opts.agentName] — The agent this process runs. Optional
|
|
42
|
+
* (agents are also discovered from your spans), but naming it here makes the
|
|
43
|
+
* agent and its heartbeat visible in the fleet before it handles a request.
|
|
44
|
+
* @param {number} [opts.heartbeatIntervalMs=60000] — Liveness beat for every
|
|
45
|
+
* agent this tracer has seen. 0 disables. The interval is sent with each
|
|
46
|
+
* beat so the server sizes THIS agent's down window to it rather than
|
|
47
|
+
* applying one global threshold to the whole fleet.
|
|
48
|
+
* @param {boolean} [opts.heartbeatMetrics=true] — Ship process metrics (RSS,
|
|
49
|
+
* event-loop lag, error rate) with each beat, so the fleet can show
|
|
50
|
+
* `degraded` before `down`.
|
|
51
|
+
* @param {boolean} [opts.propagate=true] — Inject W3C traceparent/baggage into
|
|
52
|
+
* outbound fetch/http calls made inside a span, so agents in other processes
|
|
53
|
+
* join the same trace and the topology edge forms. See propagation.js.
|
|
22
54
|
*/
|
|
23
|
-
constructor(client, {
|
|
55
|
+
constructor(client, {
|
|
56
|
+
projectName, flushIntervalMs = 5000, batchSize = 50,
|
|
57
|
+
agentName = null, heartbeatIntervalMs = 60_000, heartbeatMetrics = true,
|
|
58
|
+
propagate = true,
|
|
59
|
+
} = {}) {
|
|
24
60
|
this.#client = client;
|
|
25
61
|
this.#projectName = projectName;
|
|
26
|
-
this.#config = { flushIntervalMs, batchSize };
|
|
62
|
+
this.#config = { flushIntervalMs, batchSize, agentName, heartbeatIntervalMs, heartbeatMetrics };
|
|
63
|
+
if (agentName) this.#knownAgents.add(agentName);
|
|
27
64
|
this.#flushInterval = setInterval(() => this.flush().catch(() => {}), flushIntervalMs);
|
|
28
65
|
// Prevent the interval from keeping the process alive
|
|
29
66
|
if (this.#flushInterval.unref) this.#flushInterval.unref();
|
|
67
|
+
|
|
68
|
+
if (propagate) {
|
|
69
|
+
setDefaultProject(projectName);
|
|
70
|
+
if (client?._baseUrl) registerInternalOrigin(client._baseUrl);
|
|
71
|
+
autoInstrument();
|
|
72
|
+
}
|
|
73
|
+
if (heartbeatIntervalMs > 0 && typeof this.#client?.reportHeartbeat === 'function') {
|
|
74
|
+
this.#startHeartbeat(heartbeatIntervalMs);
|
|
75
|
+
}
|
|
76
|
+
}
|
|
77
|
+
|
|
78
|
+
/** Agents this tracer has seen (configured, or observed in a span). */
|
|
79
|
+
get knownAgents() { return [...this.#knownAgents]; }
|
|
80
|
+
|
|
81
|
+
#startHeartbeat(intervalMs) {
|
|
82
|
+
// Heartbeats did not exist in this SDK at all, so a Node agent could never
|
|
83
|
+
// report liveness: the fleet had to infer health from span traffic, which
|
|
84
|
+
// makes a quiet agent indistinguishable from a dead one. On by default, and
|
|
85
|
+
// the first beat is immediate so a process shows up at start-up rather than
|
|
86
|
+
// one interval later.
|
|
87
|
+
const beat = () => {
|
|
88
|
+
for (const agentName of this.#knownAgents) {
|
|
89
|
+
this.#client.reportHeartbeat({
|
|
90
|
+
agentName,
|
|
91
|
+
status: 'up',
|
|
92
|
+
projectName: this.#projectName,
|
|
93
|
+
heartbeatIntervalS: Math.round(intervalMs / 1000),
|
|
94
|
+
metrics: this.#config.heartbeatMetrics ? this.collectMetrics() : undefined,
|
|
95
|
+
}).catch(() => {}); // a beat must never take down the agent
|
|
96
|
+
}
|
|
97
|
+
};
|
|
98
|
+
setTimeout(beat, 0).unref?.();
|
|
99
|
+
this.#heartbeatInterval = setInterval(beat, intervalMs);
|
|
100
|
+
if (this.#heartbeatInterval.unref) this.#heartbeatInterval.unref();
|
|
30
101
|
}
|
|
31
102
|
|
|
32
103
|
/**
|
|
33
|
-
*
|
|
104
|
+
* Process metrics to ship with a heartbeat. The server turns a
|
|
105
|
+
* struggling-but-alive process into `degraded`, so the fleet shows trouble
|
|
106
|
+
* BEFORE the agent stops answering. Only keys the server whitelists are sent.
|
|
107
|
+
*/
|
|
108
|
+
collectMetrics() {
|
|
109
|
+
const m = {
|
|
110
|
+
uptime_s: Math.round((Date.now() - this.#startedAt) / 100) / 10,
|
|
111
|
+
runtime: `node ${process.versions?.node || '?'}`,
|
|
112
|
+
};
|
|
113
|
+
const spans = this.#spansSinceBeat;
|
|
114
|
+
const errors = this.#errorsSinceBeat;
|
|
115
|
+
this.#spansSinceBeat = 0;
|
|
116
|
+
this.#errorsSinceBeat = 0;
|
|
117
|
+
m.spans_since_last = spans;
|
|
118
|
+
m.errors_since_last = errors;
|
|
119
|
+
if (spans > 0) m.error_rate = Math.round((errors / spans) * 1000) / 1000;
|
|
120
|
+
try {
|
|
121
|
+
m.rss_mb = Math.round((process.memoryUsage().rss / (1024 * 1024)) * 10) / 10;
|
|
122
|
+
m.heap_mb = Math.round((process.memoryUsage().heapUsed / (1024 * 1024)) * 10) / 10;
|
|
123
|
+
} catch { /* not Node — skip */ }
|
|
124
|
+
try {
|
|
125
|
+
// Event-loop lag: how late a zero-delay timer actually ran. This is the
|
|
126
|
+
// signal that separates "busy" from "wedged" in a Node agent.
|
|
127
|
+
const lag = this.#lastLag;
|
|
128
|
+
if (typeof lag === 'number') m.event_loop_lag_ms = Math.round(lag * 100) / 100;
|
|
129
|
+
const t0 = process.hrtime.bigint();
|
|
130
|
+
setTimeout(() => {
|
|
131
|
+
this.#lastLag = Number(process.hrtime.bigint() - t0) / 1e6;
|
|
132
|
+
}, 0).unref?.();
|
|
133
|
+
} catch { /* skip */ }
|
|
134
|
+
try {
|
|
135
|
+
// CPU is a RATE, so it needs a baseline and a window wide enough to
|
|
136
|
+
// divide by. The first beat fires immediately at start-up, where the
|
|
137
|
+
// window is ~0ms and cumulative CPU is not: dividing them reported
|
|
138
|
+
// 1371% and, because a high cpu_pct is one of the server's degraded
|
|
139
|
+
// signals, marked a perfectly healthy agent degraded the moment it
|
|
140
|
+
// booted. So: no baseline or no real window means no number.
|
|
141
|
+
const now = Date.now();
|
|
142
|
+
const prev = this.#lastCpu;
|
|
143
|
+
const prevAt = this.#lastCpuAt;
|
|
144
|
+
this.#lastCpu = process.cpuUsage();
|
|
145
|
+
this.#lastCpuAt = now;
|
|
146
|
+
if (prev && prevAt && now - prevAt >= 500) {
|
|
147
|
+
const usage = process.cpuUsage(prev);
|
|
148
|
+
// Normalised by core count, so 100 means "this box is saturated"
|
|
149
|
+
// rather than "one thread is busy" — a 16-core host legitimately
|
|
150
|
+
// reports 800% per-core for work that is using half the machine.
|
|
151
|
+
const cores = Math.max(1, availableParallelism());
|
|
152
|
+
const windowUs = (now - prevAt) * 1000 * cores;
|
|
153
|
+
const pct = ((usage.user + usage.system) / windowUs) * 100;
|
|
154
|
+
m.cpu_pct = Math.min(100, Math.max(0, Math.round(pct * 10) / 10));
|
|
155
|
+
}
|
|
156
|
+
} catch { /* skip */ }
|
|
157
|
+
return m;
|
|
158
|
+
}
|
|
159
|
+
|
|
160
|
+
/**
|
|
161
|
+
* Start a new trace — or CONTINUE the caller's, when this process is serving
|
|
162
|
+
* a request that carried trace context (expressMiddleware, fastifyPlugin, or
|
|
163
|
+
* an explicit `withRemoteContext(headers, fn)`). Without continuation every
|
|
164
|
+
* service began its own trace and the graph showed disconnected agents even
|
|
165
|
+
* with headers flowing correctly.
|
|
166
|
+
*
|
|
34
167
|
* @param {string} name — Human-readable trace name
|
|
35
168
|
* @param {object} [opts]
|
|
36
169
|
* @param {string} [opts.traceId] — Provide your own trace ID, or one is generated
|
|
170
|
+
* @param {object} [opts.headers] — Inbound request headers, to continue their trace
|
|
37
171
|
* @returns {Trace}
|
|
38
172
|
*/
|
|
39
173
|
trace(name, opts = {}) {
|
|
40
174
|
return new Trace(this, name, opts);
|
|
41
175
|
}
|
|
42
176
|
|
|
177
|
+
/**
|
|
178
|
+
* Run `fn` inside a trace, with the trace active for everything it awaits.
|
|
179
|
+
* The scoped form — prefer it when you can:
|
|
180
|
+
*
|
|
181
|
+
* await tracer.withTrace('handle order', async (t) => {
|
|
182
|
+
* await t.span('RouterAgent').run(async () => { … })
|
|
183
|
+
* })
|
|
184
|
+
*/
|
|
185
|
+
async withTrace(name, fn, opts = {}) {
|
|
186
|
+
const t = this.trace(name, opts);
|
|
187
|
+
return traceStorage.run({ trace_id: t.traceId, name }, () => fn(t));
|
|
188
|
+
}
|
|
189
|
+
|
|
43
190
|
/** @internal — called by Span.end() to enqueue completed spans */
|
|
44
191
|
_enqueueSpan(span) {
|
|
192
|
+
if (span?.agent_name) this.#knownAgents.add(span.agent_name);
|
|
193
|
+
this.#spansSinceBeat++;
|
|
194
|
+
if (span?.status === 'error') this.#errorsSinceBeat++;
|
|
45
195
|
this.#spanBuffer.push(span);
|
|
46
196
|
if (this.#spanBuffer.length >= this.#config.batchSize) {
|
|
47
197
|
this.flush().catch(() => {});
|
|
@@ -85,7 +235,8 @@ export class CiphyrsTracer {
|
|
|
85
235
|
{ name: this.#projectName },
|
|
86
236
|
{ trace_id: traceId, name: traceSpans[0]?.trace_name || traceId },
|
|
87
237
|
traceSpans,
|
|
88
|
-
).catch(err =>
|
|
238
|
+
).catch(err => this.#requeue(this.#spanBuffer, traceSpans,
|
|
239
|
+
`trace ${traceId}`, err)),
|
|
89
240
|
);
|
|
90
241
|
}
|
|
91
242
|
}
|
|
@@ -95,19 +246,53 @@ export class CiphyrsTracer {
|
|
|
95
246
|
this.#client._request(`${this.#client._baseUrl}/v1/observe/metrics`, {
|
|
96
247
|
method: 'POST',
|
|
97
248
|
body: { metrics },
|
|
98
|
-
}).catch(err =>
|
|
249
|
+
}).catch(err => this.#requeue(this.#metricBuffer, metrics, 'metrics', err)),
|
|
99
250
|
);
|
|
100
251
|
}
|
|
101
252
|
|
|
102
253
|
await Promise.allSettled(promises);
|
|
103
254
|
}
|
|
104
255
|
|
|
256
|
+
/**
|
|
257
|
+
* Put a failed batch back rather than dropping it.
|
|
258
|
+
*
|
|
259
|
+
* The buffers are spliced empty BEFORE the send, so the previous `.catch`
|
|
260
|
+
* that only logged meant every failure the client's retries could not
|
|
261
|
+
* absorb destroyed that batch for good. The customer saw one line of
|
|
262
|
+
* console noise and a permanent hole in their traces.
|
|
263
|
+
*
|
|
264
|
+
* sdk-node's client.js already did it this way; this brings the published
|
|
265
|
+
* SDK in line. APPEND rather than prepend: spans are posted one trace at a
|
|
266
|
+
* time, so with traces A and B both failing, prepending would leave the
|
|
267
|
+
* buffer holding B before A and hand a recovered collector a reversed
|
|
268
|
+
* timeline.
|
|
269
|
+
*/
|
|
270
|
+
#requeue(buffer, items, what, err) {
|
|
271
|
+
buffer.push(...items);
|
|
272
|
+
// Bounded, or an unreachable collector grows this without limit inside
|
|
273
|
+
// the customer's process. Oldest go first: during an incident the recent
|
|
274
|
+
// spans are the ones worth keeping.
|
|
275
|
+
const cap = Math.max(this.#config.batchSize * 20, 1000);
|
|
276
|
+
let dropped = 0;
|
|
277
|
+
if (buffer.length > cap) {
|
|
278
|
+
dropped = buffer.length - cap;
|
|
279
|
+
buffer.splice(0, dropped);
|
|
280
|
+
}
|
|
281
|
+
console.error(
|
|
282
|
+
`[ciphyrs] ${what}: flush failed (${err.message}); `
|
|
283
|
+
+ (dropped
|
|
284
|
+
? `buffer full, dropped ${dropped} oldest item(s)`
|
|
285
|
+
: `${items.length} item(s) re-queued for the next flush`),
|
|
286
|
+
);
|
|
287
|
+
}
|
|
288
|
+
|
|
105
289
|
/**
|
|
106
290
|
* Stop the auto-flush timer and flush any remaining data.
|
|
107
291
|
* Always call this before process exit to avoid data loss.
|
|
108
292
|
*/
|
|
109
293
|
async shutdown() {
|
|
110
294
|
clearInterval(this.#flushInterval);
|
|
295
|
+
if (this.#heartbeatInterval) clearInterval(this.#heartbeatInterval);
|
|
111
296
|
await this.flush();
|
|
112
297
|
}
|
|
113
298
|
}
|
|
@@ -121,16 +306,41 @@ export class Trace {
|
|
|
121
306
|
#traceId;
|
|
122
307
|
#name;
|
|
123
308
|
#startTime;
|
|
309
|
+
#remoteParentSpanId;
|
|
310
|
+
#remoteAgent;
|
|
311
|
+
#ownSpanIds = new Set();
|
|
124
312
|
|
|
125
313
|
constructor(tracer, name, opts = {}) {
|
|
126
314
|
this.#tracer = tracer;
|
|
127
|
-
|
|
315
|
+
// Continue the caller's trace when one is active for this async context
|
|
316
|
+
// (or was handed in as `opts.headers`), unless an explicit traceId wins.
|
|
317
|
+
const remote = opts.headers ? extract(opts.headers) : remoteParent();
|
|
318
|
+
this.#traceId = opts.traceId || remote?.trace_id || newTraceId();
|
|
319
|
+
this.#remoteParentSpanId = (!opts.traceId && remote?.span_id) || null;
|
|
320
|
+
this.#remoteAgent = remote?.agent_name || null;
|
|
128
321
|
this.#name = name;
|
|
129
322
|
this.#startTime = Date.now();
|
|
130
323
|
}
|
|
131
324
|
|
|
132
325
|
get traceId() { return this.#traceId; }
|
|
133
326
|
get name() { return this.#name; }
|
|
327
|
+
/** True when this trace continues one started by another agent/process. */
|
|
328
|
+
get isContinuation() { return this.#remoteParentSpanId !== null; }
|
|
329
|
+
/** The agent that called us, when the inbound request named one. */
|
|
330
|
+
get callerAgent() { return this.#remoteAgent; }
|
|
331
|
+
/** @internal — Span.end() reports back so sibling spans can nest correctly. */
|
|
332
|
+
_ownSpan(id) { this.#ownSpanIds.add(id); }
|
|
333
|
+
/** @internal — the parent a new span should use when none was given. */
|
|
334
|
+
_defaultParent() {
|
|
335
|
+
const ambient = spanStorage.getStore();
|
|
336
|
+
if (ambient?.span_id && this.#ownSpanIds.has(ambient.span_id)) return ambient.span_id;
|
|
337
|
+
return this.#remoteParentSpanId;
|
|
338
|
+
}
|
|
339
|
+
|
|
340
|
+
/** Run `fn` with this trace active for everything it awaits. */
|
|
341
|
+
run(fn) {
|
|
342
|
+
return traceStorage.run({ trace_id: this.#traceId, name: this.#name }, () => fn(this));
|
|
343
|
+
}
|
|
134
344
|
|
|
135
345
|
/**
|
|
136
346
|
* Create a new span within this trace.
|
|
@@ -142,10 +352,20 @@ export class Trace {
|
|
|
142
352
|
* @returns {Span}
|
|
143
353
|
*/
|
|
144
354
|
span(name, opts = {}) {
|
|
145
|
-
|
|
355
|
+
const parentSpanId = opts.parentSpanId !== undefined ? opts.parentSpanId : this._defaultParent();
|
|
356
|
+
const s = new Span(this.#tracer, this.#traceId, this.#name, name, { ...opts, parentSpanId });
|
|
357
|
+
this._ownSpan(s.spanId);
|
|
358
|
+
// Auto-enter so the SDK's long-standing style nests without a callback:
|
|
359
|
+
// const outer = t.span('Router'); const inner = t.span('Billing');
|
|
360
|
+
// `inner` is a child of `outer`, and an HTTP call made between them
|
|
361
|
+
// carries `outer`'s (then `inner`'s) trace headers. end() unwinds.
|
|
362
|
+
// Pass `{ scoped: true }` to opt out and manage context with run().
|
|
363
|
+
if (opts.scoped !== true) s.enter();
|
|
364
|
+
return s;
|
|
146
365
|
}
|
|
147
366
|
}
|
|
148
367
|
|
|
368
|
+
|
|
149
369
|
// ─────────────────────────────────────────────────────────────────────────────
|
|
150
370
|
// Span — a single timed operation within a trace
|
|
151
371
|
// ─────────────────────────────────────────────────────────────────────────────
|
|
@@ -154,12 +374,15 @@ export class Span {
|
|
|
154
374
|
#tracer;
|
|
155
375
|
#data;
|
|
156
376
|
#startTime;
|
|
377
|
+
#ended = false;
|
|
378
|
+
#entered = false;
|
|
379
|
+
#previousStore = undefined;
|
|
157
380
|
|
|
158
381
|
constructor(tracer, traceId, traceName, name, opts = {}) {
|
|
159
382
|
this.#tracer = tracer;
|
|
160
383
|
this.#startTime = Date.now();
|
|
161
384
|
this.#data = {
|
|
162
|
-
span_id:
|
|
385
|
+
span_id: newSpanId(),
|
|
163
386
|
trace_id: traceId,
|
|
164
387
|
trace_name: traceName,
|
|
165
388
|
parent_span_id: opts.parentSpanId || null,
|
|
@@ -177,6 +400,49 @@ export class Span {
|
|
|
177
400
|
}
|
|
178
401
|
|
|
179
402
|
get spanId() { return this.#data.span_id; }
|
|
403
|
+
get agentName() { return this.#data.agent_name; }
|
|
404
|
+
|
|
405
|
+
/**
|
|
406
|
+
* Run `fn` with this span active, then end it — the scoped form, and the one
|
|
407
|
+
* to prefer: everything `fn` awaits sees this span as its parent, and an
|
|
408
|
+
* outbound HTTP call made inside it carries this span's trace headers.
|
|
409
|
+
* The span is ended (and marked errored) even if `fn` throws.
|
|
410
|
+
*/
|
|
411
|
+
async run(fn) {
|
|
412
|
+
// `kind` rides in the ambient store so a caller inside this span can tell
|
|
413
|
+
// WHAT span it is in, not merely that one is open. protectTool needs that
|
|
414
|
+
// distinction: the gateway matches a tool span to its gate call by span
|
|
415
|
+
// id, so sending the id of an enclosing AGENT span can never match while
|
|
416
|
+
// still telling the gateway this producer is correlated.
|
|
417
|
+
const store = { trace_id: this.#data.trace_id, span_id: this.#data.span_id, agent_name: this.#data.agent_name, kind: this.#data.kind };
|
|
418
|
+
try {
|
|
419
|
+
return await spanStorage.run(store, () => fn(this));
|
|
420
|
+
} catch (err) {
|
|
421
|
+
this.setError(err?.message || String(err));
|
|
422
|
+
throw err;
|
|
423
|
+
} finally {
|
|
424
|
+
if (!this.#ended) this.end();
|
|
425
|
+
}
|
|
426
|
+
}
|
|
427
|
+
|
|
428
|
+
/**
|
|
429
|
+
* Make this span the ambient one for the rest of the current async context,
|
|
430
|
+
* without a callback. This is what keeps the SDK's long-standing
|
|
431
|
+
* `const s = trace.span(); … s.end()` style nesting correctly — `end()`
|
|
432
|
+
* restores whatever was ambient before.
|
|
433
|
+
*/
|
|
434
|
+
enter() {
|
|
435
|
+
if (this.#entered) return this;
|
|
436
|
+
this.#entered = true;
|
|
437
|
+
this.#previousStore = spanStorage.getStore();
|
|
438
|
+
spanStorage.enterWith({
|
|
439
|
+
trace_id: this.#data.trace_id, span_id: this.#data.span_id, agent_name: this.#data.agent_name,
|
|
440
|
+
// Same store shape as run() above — the unscoped style must not be the
|
|
441
|
+
// one that leaves protectTool unable to tell a tool span from an agent span.
|
|
442
|
+
kind: this.#data.kind,
|
|
443
|
+
});
|
|
444
|
+
return this;
|
|
445
|
+
}
|
|
180
446
|
|
|
181
447
|
/** Set the input text for this span. */
|
|
182
448
|
setInput(text) { this.#data.input_text = text; return this; }
|
|
@@ -211,9 +477,12 @@ export class Span {
|
|
|
211
477
|
* @returns {this}
|
|
212
478
|
*/
|
|
213
479
|
end() {
|
|
480
|
+
if (this.#ended) return this; // double end() must not double-count
|
|
481
|
+
this.#ended = true;
|
|
214
482
|
this.#data.started_at = new Date(this.#startTime).toISOString();
|
|
215
483
|
this.#data.ended_at = new Date().toISOString();
|
|
216
484
|
this.#data.duration_ms = Date.now() - this.#startTime;
|
|
485
|
+
if (this.#entered) spanStorage.enterWith(this.#previousStore);
|
|
217
486
|
this.#tracer._enqueueSpan(this.#data);
|
|
218
487
|
return this;
|
|
219
488
|
}
|
|
@@ -0,0 +1,194 @@
|
|
|
1
|
+
/**
|
|
2
|
+
* CiphyrsTracer — buffered span and metric ingestion.
|
|
3
|
+
* Run with: npm test
|
|
4
|
+
*
|
|
5
|
+
* This is the only route spans, traces and metrics take to the platform, and
|
|
6
|
+
* its failure mode is silence: when ingestion breaks nothing errors, data just
|
|
7
|
+
* never arrives, and an empty dashboard is indistinguishable from a quiet day.
|
|
8
|
+
*
|
|
9
|
+
* The buffers are spliced empty BEFORE the send, so a `.catch` that only
|
|
10
|
+
* logged meant any failure the client's retries could not absorb destroyed
|
|
11
|
+
* that batch permanently. sdk-node's client.js already re-queued; this package
|
|
12
|
+
* did not, so the same customer got different durability from two Ciphyrs
|
|
13
|
+
* SDKs doing the same job.
|
|
14
|
+
*/
|
|
15
|
+
|
|
16
|
+
// No test in this package may open a socket — see no-network.test-helper.js.
|
|
17
|
+
// Imported per file because node --test runs each file in its own process.
|
|
18
|
+
import './no-network.test-helper.js'
|
|
19
|
+
import { describe, it, beforeEach, afterEach } from 'node:test'
|
|
20
|
+
import assert from 'node:assert/strict'
|
|
21
|
+
import { CiphyrsTracer } from './tracer.js'
|
|
22
|
+
|
|
23
|
+
// console.error is the delivery mechanism for "your telemetry is not arriving",
|
|
24
|
+
// so it is asserted on rather than silenced.
|
|
25
|
+
let logged
|
|
26
|
+
let realError
|
|
27
|
+
beforeEach(() => { logged = []; realError = console.error; console.error = (m) => logged.push(String(m)) })
|
|
28
|
+
afterEach(() => { console.error = realError })
|
|
29
|
+
|
|
30
|
+
function fakeClient({ failIngest = false, failMetrics = false } = {}) {
|
|
31
|
+
const calls = { ingest: [], metrics: [] }
|
|
32
|
+
return {
|
|
33
|
+
calls,
|
|
34
|
+
_baseUrl: 'https://api',
|
|
35
|
+
trace: {
|
|
36
|
+
ingest: async (project, trace, spans) => {
|
|
37
|
+
calls.ingest.push({ trace, spans })
|
|
38
|
+
if (failIngest) throw new Error('collector unreachable')
|
|
39
|
+
return { ok: true }
|
|
40
|
+
},
|
|
41
|
+
},
|
|
42
|
+
_request: async (url, opts) => {
|
|
43
|
+
calls.metrics.push(opts.body)
|
|
44
|
+
if (failMetrics) throw new Error('collector unreachable')
|
|
45
|
+
return { ok: true }
|
|
46
|
+
},
|
|
47
|
+
}
|
|
48
|
+
}
|
|
49
|
+
|
|
50
|
+
const tracerFor = (client, opts = {}) =>
|
|
51
|
+
// flushIntervalMs is large so the timer never races an assertion; every
|
|
52
|
+
// flush below is explicit.
|
|
53
|
+
new CiphyrsTracer(client, { projectName: 'billing', flushIntervalMs: 3_600_000, ...opts })
|
|
54
|
+
|
|
55
|
+
// Reaches the private #spanBuffer the only way a test legitimately can:
|
|
56
|
+
// enqueue, flush, and observe what a subsequent flush still tries to send.
|
|
57
|
+
const pendingSpanCount = async (tracer, client) => {
|
|
58
|
+
const before = client.calls.ingest.length
|
|
59
|
+
await tracer.flush()
|
|
60
|
+
return client.calls.ingest.slice(before).reduce((n, c) => n + c.spans.length, 0)
|
|
61
|
+
}
|
|
62
|
+
|
|
63
|
+
describe('CiphyrsTracer — a successful flush', () => {
|
|
64
|
+
it('sends buffered spans grouped into one call per trace', async () => {
|
|
65
|
+
const client = fakeClient()
|
|
66
|
+
const t = tracerFor(client)
|
|
67
|
+
t._enqueueSpan({ trace_id: 'A', name: 'a1' })
|
|
68
|
+
t._enqueueSpan({ trace_id: 'A', name: 'a2' })
|
|
69
|
+
t._enqueueSpan({ trace_id: 'B', name: 'b1' })
|
|
70
|
+
await t.flush()
|
|
71
|
+
assert.equal(client.calls.ingest.length, 2, 'traces must not be merged')
|
|
72
|
+
assert.deepEqual(client.calls.ingest.map(c => c.spans.length), [2, 1])
|
|
73
|
+
})
|
|
74
|
+
|
|
75
|
+
it('sends buffered metrics', async () => {
|
|
76
|
+
const client = fakeClient()
|
|
77
|
+
const t = tracerFor(client)
|
|
78
|
+
t.emitMetric('token_cost', 0.42, { unit: 'usd' })
|
|
79
|
+
await t.flush()
|
|
80
|
+
assert.equal(client.calls.metrics[0].metrics[0].name, 'token_cost')
|
|
81
|
+
})
|
|
82
|
+
|
|
83
|
+
it('empties the buffers, so nothing is sent twice', async () => {
|
|
84
|
+
const client = fakeClient()
|
|
85
|
+
const t = tracerFor(client)
|
|
86
|
+
t._enqueueSpan({ trace_id: 'A', name: 'a1' })
|
|
87
|
+
t.emitMetric('m', 1)
|
|
88
|
+
await t.flush()
|
|
89
|
+
await t.flush()
|
|
90
|
+
assert.equal(client.calls.ingest.length, 1)
|
|
91
|
+
assert.equal(client.calls.metrics.length, 1)
|
|
92
|
+
})
|
|
93
|
+
|
|
94
|
+
it('does nothing when there is nothing buffered', async () => {
|
|
95
|
+
const client = fakeClient()
|
|
96
|
+
await tracerFor(client).flush()
|
|
97
|
+
assert.equal(client.calls.ingest.length + client.calls.metrics.length, 0)
|
|
98
|
+
})
|
|
99
|
+
})
|
|
100
|
+
|
|
101
|
+
describe('CiphyrsTracer — a failed flush must not destroy the batch', () => {
|
|
102
|
+
it('re-queues spans and says so', async () => {
|
|
103
|
+
const client = fakeClient({ failIngest: true })
|
|
104
|
+
const t = tracerFor(client)
|
|
105
|
+
t._enqueueSpan({ trace_id: 'A', name: 'a1' })
|
|
106
|
+
await t.flush()
|
|
107
|
+
assert.match(logged.join('\n'), /re-queued/)
|
|
108
|
+
assert.equal(await pendingSpanCount(t, client), 1, 'the span was destroyed')
|
|
109
|
+
})
|
|
110
|
+
|
|
111
|
+
it('re-queues metrics and says so', async () => {
|
|
112
|
+
const client = fakeClient({ failMetrics: true })
|
|
113
|
+
const t = tracerFor(client)
|
|
114
|
+
t.emitMetric('token_cost', 0.42)
|
|
115
|
+
await t.flush()
|
|
116
|
+
const before = client.calls.metrics.length
|
|
117
|
+
await t.flush()
|
|
118
|
+
assert.ok(client.calls.metrics.length > before, 'the metric was destroyed')
|
|
119
|
+
assert.equal(client.calls.metrics.at(-1).metrics[0].name, 'token_cost')
|
|
120
|
+
})
|
|
121
|
+
|
|
122
|
+
it('delivers the batch once the collector comes back', async () => {
|
|
123
|
+
// Re-queueing is only worth anything if a later attempt picks it up.
|
|
124
|
+
const client = fakeClient({ failIngest: true })
|
|
125
|
+
const t = tracerFor(client)
|
|
126
|
+
t._enqueueSpan({ trace_id: 'A', name: 'a1' })
|
|
127
|
+
await t.flush()
|
|
128
|
+
client.trace.ingest = async (p, trace, spans) => {
|
|
129
|
+
client.calls.ingest.push({ trace, spans }); return { ok: true }
|
|
130
|
+
}
|
|
131
|
+
await t.flush()
|
|
132
|
+
assert.equal(client.calls.ingest.at(-1).spans[0].name, 'a1')
|
|
133
|
+
// ...and is then gone, not resent forever.
|
|
134
|
+
const n = client.calls.ingest.length
|
|
135
|
+
await t.flush()
|
|
136
|
+
assert.equal(client.calls.ingest.length, n)
|
|
137
|
+
})
|
|
138
|
+
|
|
139
|
+
it('keeps two failed traces in the order they were produced', async () => {
|
|
140
|
+
// Spans post one trace at a time, so the second failure lands on top of
|
|
141
|
+
// the first — the one case where re-queueing at the front would silently
|
|
142
|
+
// reverse the timeline.
|
|
143
|
+
const client = fakeClient({ failIngest: true })
|
|
144
|
+
const t = tracerFor(client)
|
|
145
|
+
t._enqueueSpan({ trace_id: 'A', name: 'a1' })
|
|
146
|
+
t._enqueueSpan({ trace_id: 'B', name: 'b1' })
|
|
147
|
+
await t.flush()
|
|
148
|
+
const before = client.calls.ingest.length
|
|
149
|
+
await t.flush()
|
|
150
|
+
const names = client.calls.ingest.slice(before).flatMap(c => c.spans.map(s => s.name))
|
|
151
|
+
assert.deepEqual(names, ['a1', 'b1'])
|
|
152
|
+
})
|
|
153
|
+
|
|
154
|
+
it('bounds the re-queue so an outage cannot exhaust memory', async () => {
|
|
155
|
+
// A tracing SDK that OOMs the process it observes is worse than one that
|
|
156
|
+
// drops telemetry — but it has to say what it dropped.
|
|
157
|
+
const client = fakeClient({ failIngest: true })
|
|
158
|
+
const t = tracerFor(client, { batchSize: 5 }) // cap = max(5*20, 1000) = 1000
|
|
159
|
+
for (let i = 0; i < 1200; i++) t._enqueueSpan({ trace_id: 'A', name: `s${i}` })
|
|
160
|
+
await t.flush()
|
|
161
|
+
assert.match(logged.join('\n'), /dropped \d+ oldest/)
|
|
162
|
+
assert.ok(await pendingSpanCount(t, client) <= 1000)
|
|
163
|
+
})
|
|
164
|
+
|
|
165
|
+
it('drops the oldest, keeping the most recent spans', async () => {
|
|
166
|
+
const client = fakeClient({ failIngest: true })
|
|
167
|
+
const t = tracerFor(client, { batchSize: 5 })
|
|
168
|
+
for (let i = 0; i < 1200; i++) t._enqueueSpan({ trace_id: 'A', name: `s${i}` })
|
|
169
|
+
await t.flush()
|
|
170
|
+
const before = client.calls.ingest.length
|
|
171
|
+
await t.flush()
|
|
172
|
+
const names = client.calls.ingest.slice(before).flatMap(c => c.spans.map(s => s.name))
|
|
173
|
+
assert.equal(names.at(-1), 's1199', 'threw away the newest instead of the oldest')
|
|
174
|
+
assert.ok(!names.includes('s0'))
|
|
175
|
+
})
|
|
176
|
+
|
|
177
|
+
it('a metrics failure does not take the spans down with it', async () => {
|
|
178
|
+
const client = fakeClient({ failMetrics: true })
|
|
179
|
+
const t = tracerFor(client)
|
|
180
|
+
t._enqueueSpan({ trace_id: 'A', name: 'a1' })
|
|
181
|
+
t.emitMetric('m', 1)
|
|
182
|
+
await t.flush()
|
|
183
|
+
assert.equal(client.calls.ingest.length, 1, 'spans were sent')
|
|
184
|
+
assert.equal(await pendingSpanCount(t, client), 0, 'spans were needlessly re-queued')
|
|
185
|
+
})
|
|
186
|
+
|
|
187
|
+
it('never rejects — observability must not crash the observed process', async () => {
|
|
188
|
+
const client = fakeClient({ failIngest: true, failMetrics: true })
|
|
189
|
+
const t = tracerFor(client)
|
|
190
|
+
t._enqueueSpan({ trace_id: 'A', name: 'a1' })
|
|
191
|
+
t.emitMetric('m', 1)
|
|
192
|
+
await t.flush() // must resolve
|
|
193
|
+
})
|
|
194
|
+
})
|