@ciphyrshq/sdk 2.6.0 → 3.0.0

This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
package/src/tracer.js CHANGED
@@ -1,4 +1,15 @@
1
- import { randomUUID } from 'node:crypto';
1
+ import { randomBytes } from 'node:crypto';
2
+ import { availableParallelism } from 'node:os';
3
+ import { spanStorage, traceStorage } from './context.js';
4
+ import { extract, registerInternalOrigin, setDefaultProject, autoInstrument } from './propagation.js';
5
+ import { remoteParent } from './context.js';
6
+
7
+ // W3C-shaped ids (32 / 16 hex). `traceparent` accepts nothing else, and
8
+ // generating them in that shape is what lets the header carry the REAL ids
9
+ // instead of a hash of them — so an OpenTelemetry-instrumented peer links to
10
+ // this exact trace. (randomUUID's dashes made it unusable as a W3C id.)
11
+ const newTraceId = () => randomBytes(16).toString('hex');
12
+ const newSpanId = () => randomBytes(8).toString('hex');
2
13
 
3
14
  // ─────────────────────────────────────────────────────────────────────────────
4
15
  // CiphyrsTracer — buffered span/metric collection with auto-flush
@@ -12,6 +23,14 @@ export class CiphyrsTracer {
12
23
  #spanBuffer = [];
13
24
  #metricBuffer = [];
14
25
  #flushInterval;
26
+ #heartbeatInterval;
27
+ #knownAgents = new Set();
28
+ #spansSinceBeat = 0;
29
+ #errorsSinceBeat = 0;
30
+ #startedAt = Date.now();
31
+ #lastCpu = null;
32
+ #lastCpuAt = null;
33
+ #lastLag = null;
15
34
 
16
35
  /**
17
36
  * @param {import('./client.js').CiphyrsClient} client
@@ -19,29 +38,160 @@ export class CiphyrsTracer {
19
38
  * @param {string} opts.projectName — Project name for grouping traces
20
39
  * @param {number} [opts.flushIntervalMs=5000] — Auto-flush interval in ms
21
40
  * @param {number} [opts.batchSize=50] — Flush when buffer reaches this size
41
+ * @param {string} [opts.agentName] — The agent this process runs. Optional
42
+ * (agents are also discovered from your spans), but naming it here makes the
43
+ * agent and its heartbeat visible in the fleet before it handles a request.
44
+ * @param {number} [opts.heartbeatIntervalMs=60000] — Liveness beat for every
45
+ * agent this tracer has seen. 0 disables. The interval is sent with each
46
+ * beat so the server sizes THIS agent's down window to it rather than
47
+ * applying one global threshold to the whole fleet.
48
+ * @param {boolean} [opts.heartbeatMetrics=true] — Ship process metrics (RSS,
49
+ * event-loop lag, error rate) with each beat, so the fleet can show
50
+ * `degraded` before `down`.
51
+ * @param {boolean} [opts.propagate=true] — Inject W3C traceparent/baggage into
52
+ * outbound fetch/http calls made inside a span, so agents in other processes
53
+ * join the same trace and the topology edge forms. See propagation.js.
22
54
  */
23
- constructor(client, { projectName, flushIntervalMs = 5000, batchSize = 50 } = {}) {
55
+ constructor(client, {
56
+ projectName, flushIntervalMs = 5000, batchSize = 50,
57
+ agentName = null, heartbeatIntervalMs = 60_000, heartbeatMetrics = true,
58
+ propagate = true,
59
+ } = {}) {
24
60
  this.#client = client;
25
61
  this.#projectName = projectName;
26
- this.#config = { flushIntervalMs, batchSize };
62
+ this.#config = { flushIntervalMs, batchSize, agentName, heartbeatIntervalMs, heartbeatMetrics };
63
+ if (agentName) this.#knownAgents.add(agentName);
27
64
  this.#flushInterval = setInterval(() => this.flush().catch(() => {}), flushIntervalMs);
28
65
  // Prevent the interval from keeping the process alive
29
66
  if (this.#flushInterval.unref) this.#flushInterval.unref();
67
+
68
+ if (propagate) {
69
+ setDefaultProject(projectName);
70
+ if (client?._baseUrl) registerInternalOrigin(client._baseUrl);
71
+ autoInstrument();
72
+ }
73
+ if (heartbeatIntervalMs > 0 && typeof this.#client?.reportHeartbeat === 'function') {
74
+ this.#startHeartbeat(heartbeatIntervalMs);
75
+ }
76
+ }
77
+
78
+ /** Agents this tracer has seen (configured, or observed in a span). */
79
+ get knownAgents() { return [...this.#knownAgents]; }
80
+
81
+ #startHeartbeat(intervalMs) {
82
+ // Heartbeats did not exist in this SDK at all, so a Node agent could never
83
+ // report liveness: the fleet had to infer health from span traffic, which
84
+ // makes a quiet agent indistinguishable from a dead one. On by default, and
85
+ // the first beat is immediate so a process shows up at start-up rather than
86
+ // one interval later.
87
+ const beat = () => {
88
+ for (const agentName of this.#knownAgents) {
89
+ this.#client.reportHeartbeat({
90
+ agentName,
91
+ status: 'up',
92
+ projectName: this.#projectName,
93
+ heartbeatIntervalS: Math.round(intervalMs / 1000),
94
+ metrics: this.#config.heartbeatMetrics ? this.collectMetrics() : undefined,
95
+ }).catch(() => {}); // a beat must never take down the agent
96
+ }
97
+ };
98
+ setTimeout(beat, 0).unref?.();
99
+ this.#heartbeatInterval = setInterval(beat, intervalMs);
100
+ if (this.#heartbeatInterval.unref) this.#heartbeatInterval.unref();
30
101
  }
31
102
 
32
103
  /**
33
- * Start a new trace.
104
+ * Process metrics to ship with a heartbeat. The server turns a
105
+ * struggling-but-alive process into `degraded`, so the fleet shows trouble
106
+ * BEFORE the agent stops answering. Only keys the server whitelists are sent.
107
+ */
108
+ collectMetrics() {
109
+ const m = {
110
+ uptime_s: Math.round((Date.now() - this.#startedAt) / 100) / 10,
111
+ runtime: `node ${process.versions?.node || '?'}`,
112
+ };
113
+ const spans = this.#spansSinceBeat;
114
+ const errors = this.#errorsSinceBeat;
115
+ this.#spansSinceBeat = 0;
116
+ this.#errorsSinceBeat = 0;
117
+ m.spans_since_last = spans;
118
+ m.errors_since_last = errors;
119
+ if (spans > 0) m.error_rate = Math.round((errors / spans) * 1000) / 1000;
120
+ try {
121
+ m.rss_mb = Math.round((process.memoryUsage().rss / (1024 * 1024)) * 10) / 10;
122
+ m.heap_mb = Math.round((process.memoryUsage().heapUsed / (1024 * 1024)) * 10) / 10;
123
+ } catch { /* not Node — skip */ }
124
+ try {
125
+ // Event-loop lag: how late a zero-delay timer actually ran. This is the
126
+ // signal that separates "busy" from "wedged" in a Node agent.
127
+ const lag = this.#lastLag;
128
+ if (typeof lag === 'number') m.event_loop_lag_ms = Math.round(lag * 100) / 100;
129
+ const t0 = process.hrtime.bigint();
130
+ setTimeout(() => {
131
+ this.#lastLag = Number(process.hrtime.bigint() - t0) / 1e6;
132
+ }, 0).unref?.();
133
+ } catch { /* skip */ }
134
+ try {
135
+ // CPU is a RATE, so it needs a baseline and a window wide enough to
136
+ // divide by. The first beat fires immediately at start-up, where the
137
+ // window is ~0ms and cumulative CPU is not: dividing them reported
138
+ // 1371% and, because a high cpu_pct is one of the server's degraded
139
+ // signals, marked a perfectly healthy agent degraded the moment it
140
+ // booted. So: no baseline or no real window means no number.
141
+ const now = Date.now();
142
+ const prev = this.#lastCpu;
143
+ const prevAt = this.#lastCpuAt;
144
+ this.#lastCpu = process.cpuUsage();
145
+ this.#lastCpuAt = now;
146
+ if (prev && prevAt && now - prevAt >= 500) {
147
+ const usage = process.cpuUsage(prev);
148
+ // Normalised by core count, so 100 means "this box is saturated"
149
+ // rather than "one thread is busy" — a 16-core host legitimately
150
+ // reports 800% per-core for work that is using half the machine.
151
+ const cores = Math.max(1, availableParallelism());
152
+ const windowUs = (now - prevAt) * 1000 * cores;
153
+ const pct = ((usage.user + usage.system) / windowUs) * 100;
154
+ m.cpu_pct = Math.min(100, Math.max(0, Math.round(pct * 10) / 10));
155
+ }
156
+ } catch { /* skip */ }
157
+ return m;
158
+ }
159
+
160
+ /**
161
+ * Start a new trace — or CONTINUE the caller's, when this process is serving
162
+ * a request that carried trace context (expressMiddleware, fastifyPlugin, or
163
+ * an explicit `withRemoteContext(headers, fn)`). Without continuation every
164
+ * service began its own trace and the graph showed disconnected agents even
165
+ * with headers flowing correctly.
166
+ *
34
167
  * @param {string} name — Human-readable trace name
35
168
  * @param {object} [opts]
36
169
  * @param {string} [opts.traceId] — Provide your own trace ID, or one is generated
170
+ * @param {object} [opts.headers] — Inbound request headers, to continue their trace
37
171
  * @returns {Trace}
38
172
  */
39
173
  trace(name, opts = {}) {
40
174
  return new Trace(this, name, opts);
41
175
  }
42
176
 
177
+ /**
178
+ * Run `fn` inside a trace, with the trace active for everything it awaits.
179
+ * The scoped form — prefer it when you can:
180
+ *
181
+ * await tracer.withTrace('handle order', async (t) => {
182
+ * await t.span('RouterAgent').run(async () => { … })
183
+ * })
184
+ */
185
+ async withTrace(name, fn, opts = {}) {
186
+ const t = this.trace(name, opts);
187
+ return traceStorage.run({ trace_id: t.traceId, name }, () => fn(t));
188
+ }
189
+
43
190
  /** @internal — called by Span.end() to enqueue completed spans */
44
191
  _enqueueSpan(span) {
192
+ if (span?.agent_name) this.#knownAgents.add(span.agent_name);
193
+ this.#spansSinceBeat++;
194
+ if (span?.status === 'error') this.#errorsSinceBeat++;
45
195
  this.#spanBuffer.push(span);
46
196
  if (this.#spanBuffer.length >= this.#config.batchSize) {
47
197
  this.flush().catch(() => {});
@@ -85,7 +235,8 @@ export class CiphyrsTracer {
85
235
  { name: this.#projectName },
86
236
  { trace_id: traceId, name: traceSpans[0]?.trace_name || traceId },
87
237
  traceSpans,
88
- ).catch(err => console.error('Ciphyrs flush error:', err.message)),
238
+ ).catch(err => this.#requeue(this.#spanBuffer, traceSpans,
239
+ `trace ${traceId}`, err)),
89
240
  );
90
241
  }
91
242
  }
@@ -95,19 +246,53 @@ export class CiphyrsTracer {
95
246
  this.#client._request(`${this.#client._baseUrl}/v1/observe/metrics`, {
96
247
  method: 'POST',
97
248
  body: { metrics },
98
- }).catch(err => console.error('Ciphyrs metrics flush error:', err.message)),
249
+ }).catch(err => this.#requeue(this.#metricBuffer, metrics, 'metrics', err)),
99
250
  );
100
251
  }
101
252
 
102
253
  await Promise.allSettled(promises);
103
254
  }
104
255
 
256
+ /**
257
+ * Put a failed batch back rather than dropping it.
258
+ *
259
+ * The buffers are spliced empty BEFORE the send, so the previous `.catch`
260
+ * that only logged meant every failure the client's retries could not
261
+ * absorb destroyed that batch for good. The customer saw one line of
262
+ * console noise and a permanent hole in their traces.
263
+ *
264
+ * sdk-node's client.js already did it this way; this brings the published
265
+ * SDK in line. APPEND rather than prepend: spans are posted one trace at a
266
+ * time, so with traces A and B both failing, prepending would leave the
267
+ * buffer holding B before A and hand a recovered collector a reversed
268
+ * timeline.
269
+ */
270
+ #requeue(buffer, items, what, err) {
271
+ buffer.push(...items);
272
+ // Bounded, or an unreachable collector grows this without limit inside
273
+ // the customer's process. Oldest go first: during an incident the recent
274
+ // spans are the ones worth keeping.
275
+ const cap = Math.max(this.#config.batchSize * 20, 1000);
276
+ let dropped = 0;
277
+ if (buffer.length > cap) {
278
+ dropped = buffer.length - cap;
279
+ buffer.splice(0, dropped);
280
+ }
281
+ console.error(
282
+ `[ciphyrs] ${what}: flush failed (${err.message}); `
283
+ + (dropped
284
+ ? `buffer full, dropped ${dropped} oldest item(s)`
285
+ : `${items.length} item(s) re-queued for the next flush`),
286
+ );
287
+ }
288
+
105
289
  /**
106
290
  * Stop the auto-flush timer and flush any remaining data.
107
291
  * Always call this before process exit to avoid data loss.
108
292
  */
109
293
  async shutdown() {
110
294
  clearInterval(this.#flushInterval);
295
+ if (this.#heartbeatInterval) clearInterval(this.#heartbeatInterval);
111
296
  await this.flush();
112
297
  }
113
298
  }
@@ -121,16 +306,41 @@ export class Trace {
121
306
  #traceId;
122
307
  #name;
123
308
  #startTime;
309
+ #remoteParentSpanId;
310
+ #remoteAgent;
311
+ #ownSpanIds = new Set();
124
312
 
125
313
  constructor(tracer, name, opts = {}) {
126
314
  this.#tracer = tracer;
127
- this.#traceId = opts.traceId || randomUUID();
315
+ // Continue the caller's trace when one is active for this async context
316
+ // (or was handed in as `opts.headers`), unless an explicit traceId wins.
317
+ const remote = opts.headers ? extract(opts.headers) : remoteParent();
318
+ this.#traceId = opts.traceId || remote?.trace_id || newTraceId();
319
+ this.#remoteParentSpanId = (!opts.traceId && remote?.span_id) || null;
320
+ this.#remoteAgent = remote?.agent_name || null;
128
321
  this.#name = name;
129
322
  this.#startTime = Date.now();
130
323
  }
131
324
 
132
325
  get traceId() { return this.#traceId; }
133
326
  get name() { return this.#name; }
327
+ /** True when this trace continues one started by another agent/process. */
328
+ get isContinuation() { return this.#remoteParentSpanId !== null; }
329
+ /** The agent that called us, when the inbound request named one. */
330
+ get callerAgent() { return this.#remoteAgent; }
331
+ /** @internal — Span.end() reports back so sibling spans can nest correctly. */
332
+ _ownSpan(id) { this.#ownSpanIds.add(id); }
333
+ /** @internal — the parent a new span should use when none was given. */
334
+ _defaultParent() {
335
+ const ambient = spanStorage.getStore();
336
+ if (ambient?.span_id && this.#ownSpanIds.has(ambient.span_id)) return ambient.span_id;
337
+ return this.#remoteParentSpanId;
338
+ }
339
+
340
+ /** Run `fn` with this trace active for everything it awaits. */
341
+ run(fn) {
342
+ return traceStorage.run({ trace_id: this.#traceId, name: this.#name }, () => fn(this));
343
+ }
134
344
 
135
345
  /**
136
346
  * Create a new span within this trace.
@@ -142,10 +352,20 @@ export class Trace {
142
352
  * @returns {Span}
143
353
  */
144
354
  span(name, opts = {}) {
145
- return new Span(this.#tracer, this.#traceId, this.#name, name, opts);
355
+ const parentSpanId = opts.parentSpanId !== undefined ? opts.parentSpanId : this._defaultParent();
356
+ const s = new Span(this.#tracer, this.#traceId, this.#name, name, { ...opts, parentSpanId });
357
+ this._ownSpan(s.spanId);
358
+ // Auto-enter so the SDK's long-standing style nests without a callback:
359
+ // const outer = t.span('Router'); const inner = t.span('Billing');
360
+ // `inner` is a child of `outer`, and an HTTP call made between them
361
+ // carries `outer`'s (then `inner`'s) trace headers. end() unwinds.
362
+ // Pass `{ scoped: true }` to opt out and manage context with run().
363
+ if (opts.scoped !== true) s.enter();
364
+ return s;
146
365
  }
147
366
  }
148
367
 
368
+
149
369
  // ─────────────────────────────────────────────────────────────────────────────
150
370
  // Span — a single timed operation within a trace
151
371
  // ─────────────────────────────────────────────────────────────────────────────
@@ -154,12 +374,15 @@ export class Span {
154
374
  #tracer;
155
375
  #data;
156
376
  #startTime;
377
+ #ended = false;
378
+ #entered = false;
379
+ #previousStore = undefined;
157
380
 
158
381
  constructor(tracer, traceId, traceName, name, opts = {}) {
159
382
  this.#tracer = tracer;
160
383
  this.#startTime = Date.now();
161
384
  this.#data = {
162
- span_id: randomUUID(),
385
+ span_id: newSpanId(),
163
386
  trace_id: traceId,
164
387
  trace_name: traceName,
165
388
  parent_span_id: opts.parentSpanId || null,
@@ -177,6 +400,49 @@ export class Span {
177
400
  }
178
401
 
179
402
  get spanId() { return this.#data.span_id; }
403
+ get agentName() { return this.#data.agent_name; }
404
+
405
+ /**
406
+ * Run `fn` with this span active, then end it — the scoped form, and the one
407
+ * to prefer: everything `fn` awaits sees this span as its parent, and an
408
+ * outbound HTTP call made inside it carries this span's trace headers.
409
+ * The span is ended (and marked errored) even if `fn` throws.
410
+ */
411
+ async run(fn) {
412
+ // `kind` rides in the ambient store so a caller inside this span can tell
413
+ // WHAT span it is in, not merely that one is open. protectTool needs that
414
+ // distinction: the gateway matches a tool span to its gate call by span
415
+ // id, so sending the id of an enclosing AGENT span can never match while
416
+ // still telling the gateway this producer is correlated.
417
+ const store = { trace_id: this.#data.trace_id, span_id: this.#data.span_id, agent_name: this.#data.agent_name, kind: this.#data.kind };
418
+ try {
419
+ return await spanStorage.run(store, () => fn(this));
420
+ } catch (err) {
421
+ this.setError(err?.message || String(err));
422
+ throw err;
423
+ } finally {
424
+ if (!this.#ended) this.end();
425
+ }
426
+ }
427
+
428
+ /**
429
+ * Make this span the ambient one for the rest of the current async context,
430
+ * without a callback. This is what keeps the SDK's long-standing
431
+ * `const s = trace.span(); … s.end()` style nesting correctly — `end()`
432
+ * restores whatever was ambient before.
433
+ */
434
+ enter() {
435
+ if (this.#entered) return this;
436
+ this.#entered = true;
437
+ this.#previousStore = spanStorage.getStore();
438
+ spanStorage.enterWith({
439
+ trace_id: this.#data.trace_id, span_id: this.#data.span_id, agent_name: this.#data.agent_name,
440
+ // Same store shape as run() above — the unscoped style must not be the
441
+ // one that leaves protectTool unable to tell a tool span from an agent span.
442
+ kind: this.#data.kind,
443
+ });
444
+ return this;
445
+ }
180
446
 
181
447
  /** Set the input text for this span. */
182
448
  setInput(text) { this.#data.input_text = text; return this; }
@@ -211,9 +477,12 @@ export class Span {
211
477
  * @returns {this}
212
478
  */
213
479
  end() {
480
+ if (this.#ended) return this; // double end() must not double-count
481
+ this.#ended = true;
214
482
  this.#data.started_at = new Date(this.#startTime).toISOString();
215
483
  this.#data.ended_at = new Date().toISOString();
216
484
  this.#data.duration_ms = Date.now() - this.#startTime;
485
+ if (this.#entered) spanStorage.enterWith(this.#previousStore);
217
486
  this.#tracer._enqueueSpan(this.#data);
218
487
  return this;
219
488
  }
@@ -0,0 +1,194 @@
1
+ /**
2
+ * CiphyrsTracer — buffered span and metric ingestion.
3
+ * Run with: npm test
4
+ *
5
+ * This is the only route spans, traces and metrics take to the platform, and
6
+ * its failure mode is silence: when ingestion breaks nothing errors, data just
7
+ * never arrives, and an empty dashboard is indistinguishable from a quiet day.
8
+ *
9
+ * The buffers are spliced empty BEFORE the send, so a `.catch` that only
10
+ * logged meant any failure the client's retries could not absorb destroyed
11
+ * that batch permanently. sdk-node's client.js already re-queued; this package
12
+ * did not, so the same customer got different durability from two Ciphyrs
13
+ * SDKs doing the same job.
14
+ */
15
+
16
+ // No test in this package may open a socket — see no-network.test-helper.js.
17
+ // Imported per file because node --test runs each file in its own process.
18
+ import './no-network.test-helper.js'
19
+ import { describe, it, beforeEach, afterEach } from 'node:test'
20
+ import assert from 'node:assert/strict'
21
+ import { CiphyrsTracer } from './tracer.js'
22
+
23
+ // console.error is the delivery mechanism for "your telemetry is not arriving",
24
+ // so it is asserted on rather than silenced.
25
+ let logged
26
+ let realError
27
+ beforeEach(() => { logged = []; realError = console.error; console.error = (m) => logged.push(String(m)) })
28
+ afterEach(() => { console.error = realError })
29
+
30
+ function fakeClient({ failIngest = false, failMetrics = false } = {}) {
31
+ const calls = { ingest: [], metrics: [] }
32
+ return {
33
+ calls,
34
+ _baseUrl: 'https://api',
35
+ trace: {
36
+ ingest: async (project, trace, spans) => {
37
+ calls.ingest.push({ trace, spans })
38
+ if (failIngest) throw new Error('collector unreachable')
39
+ return { ok: true }
40
+ },
41
+ },
42
+ _request: async (url, opts) => {
43
+ calls.metrics.push(opts.body)
44
+ if (failMetrics) throw new Error('collector unreachable')
45
+ return { ok: true }
46
+ },
47
+ }
48
+ }
49
+
50
+ const tracerFor = (client, opts = {}) =>
51
+ // flushIntervalMs is large so the timer never races an assertion; every
52
+ // flush below is explicit.
53
+ new CiphyrsTracer(client, { projectName: 'billing', flushIntervalMs: 3_600_000, ...opts })
54
+
55
+ // Reaches the private #spanBuffer the only way a test legitimately can:
56
+ // enqueue, flush, and observe what a subsequent flush still tries to send.
57
+ const pendingSpanCount = async (tracer, client) => {
58
+ const before = client.calls.ingest.length
59
+ await tracer.flush()
60
+ return client.calls.ingest.slice(before).reduce((n, c) => n + c.spans.length, 0)
61
+ }
62
+
63
+ describe('CiphyrsTracer — a successful flush', () => {
64
+ it('sends buffered spans grouped into one call per trace', async () => {
65
+ const client = fakeClient()
66
+ const t = tracerFor(client)
67
+ t._enqueueSpan({ trace_id: 'A', name: 'a1' })
68
+ t._enqueueSpan({ trace_id: 'A', name: 'a2' })
69
+ t._enqueueSpan({ trace_id: 'B', name: 'b1' })
70
+ await t.flush()
71
+ assert.equal(client.calls.ingest.length, 2, 'traces must not be merged')
72
+ assert.deepEqual(client.calls.ingest.map(c => c.spans.length), [2, 1])
73
+ })
74
+
75
+ it('sends buffered metrics', async () => {
76
+ const client = fakeClient()
77
+ const t = tracerFor(client)
78
+ t.emitMetric('token_cost', 0.42, { unit: 'usd' })
79
+ await t.flush()
80
+ assert.equal(client.calls.metrics[0].metrics[0].name, 'token_cost')
81
+ })
82
+
83
+ it('empties the buffers, so nothing is sent twice', async () => {
84
+ const client = fakeClient()
85
+ const t = tracerFor(client)
86
+ t._enqueueSpan({ trace_id: 'A', name: 'a1' })
87
+ t.emitMetric('m', 1)
88
+ await t.flush()
89
+ await t.flush()
90
+ assert.equal(client.calls.ingest.length, 1)
91
+ assert.equal(client.calls.metrics.length, 1)
92
+ })
93
+
94
+ it('does nothing when there is nothing buffered', async () => {
95
+ const client = fakeClient()
96
+ await tracerFor(client).flush()
97
+ assert.equal(client.calls.ingest.length + client.calls.metrics.length, 0)
98
+ })
99
+ })
100
+
101
+ describe('CiphyrsTracer — a failed flush must not destroy the batch', () => {
102
+ it('re-queues spans and says so', async () => {
103
+ const client = fakeClient({ failIngest: true })
104
+ const t = tracerFor(client)
105
+ t._enqueueSpan({ trace_id: 'A', name: 'a1' })
106
+ await t.flush()
107
+ assert.match(logged.join('\n'), /re-queued/)
108
+ assert.equal(await pendingSpanCount(t, client), 1, 'the span was destroyed')
109
+ })
110
+
111
+ it('re-queues metrics and says so', async () => {
112
+ const client = fakeClient({ failMetrics: true })
113
+ const t = tracerFor(client)
114
+ t.emitMetric('token_cost', 0.42)
115
+ await t.flush()
116
+ const before = client.calls.metrics.length
117
+ await t.flush()
118
+ assert.ok(client.calls.metrics.length > before, 'the metric was destroyed')
119
+ assert.equal(client.calls.metrics.at(-1).metrics[0].name, 'token_cost')
120
+ })
121
+
122
+ it('delivers the batch once the collector comes back', async () => {
123
+ // Re-queueing is only worth anything if a later attempt picks it up.
124
+ const client = fakeClient({ failIngest: true })
125
+ const t = tracerFor(client)
126
+ t._enqueueSpan({ trace_id: 'A', name: 'a1' })
127
+ await t.flush()
128
+ client.trace.ingest = async (p, trace, spans) => {
129
+ client.calls.ingest.push({ trace, spans }); return { ok: true }
130
+ }
131
+ await t.flush()
132
+ assert.equal(client.calls.ingest.at(-1).spans[0].name, 'a1')
133
+ // ...and is then gone, not resent forever.
134
+ const n = client.calls.ingest.length
135
+ await t.flush()
136
+ assert.equal(client.calls.ingest.length, n)
137
+ })
138
+
139
+ it('keeps two failed traces in the order they were produced', async () => {
140
+ // Spans post one trace at a time, so the second failure lands on top of
141
+ // the first — the one case where re-queueing at the front would silently
142
+ // reverse the timeline.
143
+ const client = fakeClient({ failIngest: true })
144
+ const t = tracerFor(client)
145
+ t._enqueueSpan({ trace_id: 'A', name: 'a1' })
146
+ t._enqueueSpan({ trace_id: 'B', name: 'b1' })
147
+ await t.flush()
148
+ const before = client.calls.ingest.length
149
+ await t.flush()
150
+ const names = client.calls.ingest.slice(before).flatMap(c => c.spans.map(s => s.name))
151
+ assert.deepEqual(names, ['a1', 'b1'])
152
+ })
153
+
154
+ it('bounds the re-queue so an outage cannot exhaust memory', async () => {
155
+ // A tracing SDK that OOMs the process it observes is worse than one that
156
+ // drops telemetry — but it has to say what it dropped.
157
+ const client = fakeClient({ failIngest: true })
158
+ const t = tracerFor(client, { batchSize: 5 }) // cap = max(5*20, 1000) = 1000
159
+ for (let i = 0; i < 1200; i++) t._enqueueSpan({ trace_id: 'A', name: `s${i}` })
160
+ await t.flush()
161
+ assert.match(logged.join('\n'), /dropped \d+ oldest/)
162
+ assert.ok(await pendingSpanCount(t, client) <= 1000)
163
+ })
164
+
165
+ it('drops the oldest, keeping the most recent spans', async () => {
166
+ const client = fakeClient({ failIngest: true })
167
+ const t = tracerFor(client, { batchSize: 5 })
168
+ for (let i = 0; i < 1200; i++) t._enqueueSpan({ trace_id: 'A', name: `s${i}` })
169
+ await t.flush()
170
+ const before = client.calls.ingest.length
171
+ await t.flush()
172
+ const names = client.calls.ingest.slice(before).flatMap(c => c.spans.map(s => s.name))
173
+ assert.equal(names.at(-1), 's1199', 'threw away the newest instead of the oldest')
174
+ assert.ok(!names.includes('s0'))
175
+ })
176
+
177
+ it('a metrics failure does not take the spans down with it', async () => {
178
+ const client = fakeClient({ failMetrics: true })
179
+ const t = tracerFor(client)
180
+ t._enqueueSpan({ trace_id: 'A', name: 'a1' })
181
+ t.emitMetric('m', 1)
182
+ await t.flush()
183
+ assert.equal(client.calls.ingest.length, 1, 'spans were sent')
184
+ assert.equal(await pendingSpanCount(t, client), 0, 'spans were needlessly re-queued')
185
+ })
186
+
187
+ it('never rejects — observability must not crash the observed process', async () => {
188
+ const client = fakeClient({ failIngest: true, failMetrics: true })
189
+ const t = tracerFor(client)
190
+ t._enqueueSpan({ trace_id: 'A', name: 'a1' })
191
+ t.emitMetric('m', 1)
192
+ await t.flush() // must resolve
193
+ })
194
+ })