@intentius/chant-lexicon-otel 0.100.0 → 0.102.0

This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
Files changed (131) hide show
  1. package/README.md +8 -6
  2. package/dist/codegen/docs.d.ts.map +1 -1
  3. package/dist/components/common.d.ts +2 -0
  4. package/dist/components/common.d.ts.map +1 -1
  5. package/dist/components/connectors.d.ts +109 -1
  6. package/dist/components/connectors.d.ts.map +1 -1
  7. package/dist/components/exporters.d.ts.map +1 -1
  8. package/dist/components/extensions.d.ts +26 -2
  9. package/dist/components/extensions.d.ts.map +1 -1
  10. package/dist/components/processors.d.ts +15 -2
  11. package/dist/components/processors.d.ts.map +1 -1
  12. package/dist/components/receivers.d.ts.map +1 -1
  13. package/dist/composites/catalog.d.ts +3 -0
  14. package/dist/composites/catalog.d.ts.map +1 -0
  15. package/dist/composites/index.d.ts +6 -0
  16. package/dist/composites/index.d.ts.map +1 -0
  17. package/dist/composites/node-agent.d.ts +111 -0
  18. package/dist/composites/node-agent.d.ts.map +1 -0
  19. package/dist/configmap.d.ts +35 -0
  20. package/dist/configmap.d.ts.map +1 -0
  21. package/dist/define.d.ts +9 -0
  22. package/dist/define.d.ts.map +1 -1
  23. package/dist/genai.d.ts +145 -2
  24. package/dist/genai.d.ts.map +1 -1
  25. package/dist/import/generator.d.ts.map +1 -1
  26. package/dist/import/parser.d.ts.map +1 -1
  27. package/dist/index.d.ts +3 -1
  28. package/dist/index.d.ts.map +1 -1
  29. package/dist/init-templates.d.ts +24 -0
  30. package/dist/init-templates.d.ts.map +1 -0
  31. package/dist/integrity.json +10 -5
  32. package/dist/lint/audit-catalog.d.ts +1 -1
  33. package/dist/lint/audit-catalog.d.ts.map +1 -1
  34. package/dist/lint/post-synth/index.d.ts.map +1 -1
  35. package/dist/lint/post-synth/otel-helpers.d.ts +32 -9
  36. package/dist/lint/post-synth/otel-helpers.d.ts.map +1 -1
  37. package/dist/lint/post-synth/otel113.d.ts +8 -0
  38. package/dist/lint/post-synth/otel113.d.ts.map +1 -0
  39. package/dist/lint/post-synth/otel114.d.ts +8 -0
  40. package/dist/lint/post-synth/otel114.d.ts.map +1 -0
  41. package/dist/lint/post-synth/otel115.d.ts +8 -0
  42. package/dist/lint/post-synth/otel115.d.ts.map +1 -0
  43. package/dist/lint/post-synth/otel116.d.ts +8 -0
  44. package/dist/lint/post-synth/otel116.d.ts.map +1 -0
  45. package/dist/lint/post-synth/otel117.d.ts +8 -0
  46. package/dist/lint/post-synth/otel117.d.ts.map +1 -0
  47. package/dist/manifest.json +1 -1
  48. package/dist/meta.json +15 -0
  49. package/dist/model.d.ts.map +1 -1
  50. package/dist/okf/index.md +8 -0
  51. package/dist/okf/rules/OTEL113.md +11 -0
  52. package/dist/okf/rules/OTEL114.md +11 -0
  53. package/dist/okf/rules/OTEL115.md +11 -0
  54. package/dist/okf/rules/OTEL116.md +11 -0
  55. package/dist/okf/rules/OTEL117.md +11 -0
  56. package/dist/okf/types/DeltaToCumulativeProcessor.md +9 -0
  57. package/dist/okf/types/K8sLeaderElectorExtension.md +9 -0
  58. package/dist/okf/types/SignalToMetricsConnector.md +9 -0
  59. package/dist/platform.d.ts +6 -2
  60. package/dist/platform.d.ts.map +1 -1
  61. package/dist/plugin.d.ts.map +1 -1
  62. package/dist/rules/otel-helpers.ts +43 -11
  63. package/dist/rules/otel113.ts +17 -0
  64. package/dist/rules/otel114.ts +17 -0
  65. package/dist/rules/otel115.ts +17 -0
  66. package/dist/rules/otel116.ts +17 -0
  67. package/dist/rules/otel117.ts +17 -0
  68. package/dist/skill-defs.d.ts +1 -1
  69. package/dist/skill-defs.d.ts.map +1 -1
  70. package/dist/skills/chant-otel.md +11 -5
  71. package/dist/topology.d.ts +5 -0
  72. package/dist/topology.d.ts.map +1 -1
  73. package/dist/validate-config.d.ts +14 -2
  74. package/dist/validate-config.d.ts.map +1 -1
  75. package/dist/validate.d.ts.map +1 -1
  76. package/package.json +3 -3
  77. package/src/codegen/docs.ts +9 -4
  78. package/src/components/common.ts +17 -0
  79. package/src/components/connectors.ts +209 -1
  80. package/src/components/exporters.ts +2 -0
  81. package/src/components/extensions.test.ts +70 -0
  82. package/src/components/extensions.ts +60 -2
  83. package/src/components/processors.test.ts +54 -0
  84. package/src/components/processors.ts +41 -2
  85. package/src/components/receivers.ts +5 -0
  86. package/src/composites/catalog.test.ts +38 -0
  87. package/src/composites/catalog.ts +81 -0
  88. package/src/composites/composites.test.ts +138 -0
  89. package/src/composites/index.ts +6 -0
  90. package/src/composites/node-agent.ts +220 -0
  91. package/src/configmap.ts +67 -0
  92. package/src/connectors.test.ts +197 -1
  93. package/src/define.ts +9 -0
  94. package/src/genai.test.ts +301 -6
  95. package/src/genai.ts +323 -8
  96. package/src/generated/lexicon-otel.json +15 -0
  97. package/src/import/roundtrip.test.ts +21 -1
  98. package/src/import/testdata/fixtures.ts +37 -5
  99. package/src/import/testdata/k8s-leader-elector.yaml +35 -0
  100. package/src/import/testdata/signaltometrics.yaml +83 -0
  101. package/src/index.ts +16 -0
  102. package/src/init-templates.test.ts +72 -0
  103. package/src/init-templates.ts +148 -0
  104. package/src/lint/audit-catalog.ts +41 -1
  105. package/src/lint/post-synth/connector-graph.test.ts +207 -0
  106. package/src/lint/post-synth/index.ts +10 -0
  107. package/src/lint/post-synth/otel-helpers.ts +43 -11
  108. package/src/lint/post-synth/otel113.ts +17 -0
  109. package/src/lint/post-synth/otel114.ts +17 -0
  110. package/src/lint/post-synth/otel115.ts +17 -0
  111. package/src/lint/post-synth/otel116.ts +17 -0
  112. package/src/lint/post-synth/otel117.ts +17 -0
  113. package/src/lint/post-synth/post-synth.test.ts +236 -1
  114. package/src/lint/post-synth/testdata/otel113-fail.yaml +23 -0
  115. package/src/lint/post-synth/testdata/otel113-pass.yaml +21 -0
  116. package/src/lint/post-synth/testdata/otel114-fail.yaml +25 -0
  117. package/src/lint/post-synth/testdata/otel114-pass.yaml +23 -0
  118. package/src/lint/post-synth/testdata/otel115-fail.yaml +29 -0
  119. package/src/lint/post-synth/testdata/otel115-pass.yaml +29 -0
  120. package/src/platform.test.ts +26 -0
  121. package/src/platform.ts +17 -13
  122. package/src/plugin.test.ts +18 -0
  123. package/src/plugin.ts +22 -0
  124. package/src/skills/chant-otel.md +11 -5
  125. package/src/testdata/genai-configured.yaml +108 -0
  126. package/src/testdata/genai-default.yaml +92 -0
  127. package/src/topology.test.ts +12 -0
  128. package/src/topology.ts +16 -0
  129. package/src/typecheck.test.ts +112 -0
  130. package/src/validate-config.ts +426 -2
  131. package/src/validate.ts +17 -0
package/src/genai.ts CHANGED
@@ -17,12 +17,27 @@
17
17
  * output tokens by model). `spanmetrics` counts spans and
18
18
  * cannot sum an attribute, which is why token usage comes from the `sum`
19
19
  * connector.
20
+ *
21
+ * With `clientMetrics: "derive"`, a `signaltometrics` connector on the same
22
+ * branch also emits the two client metrics the conventions define at the pin,
23
+ * `gen_ai.client.operation.duration` and `gen_ai.client.token.usage`, under
24
+ * their own names, attributes, units and buckets. `clientMetrics:
25
+ * "passthrough"` derives nothing and relies on the SDK to send them. Either
26
+ * way a `metrics` pipeline passes the SDK's OTLP metrics through. Without the
27
+ * option the output is what it was before the option existed.
28
+ *
29
+ * `sum` emits delta sums. The `prometheus` exporter accumulates them itself,
30
+ * but `prometheusremotewrite` drops them, and some OTLP backends store only
31
+ * cumulative data; `signaltometrics` emits delta histograms too.
32
+ * `deltaToCumulative` puts a `deltatocumulative` processor on `metrics/genai`
33
+ * for those. It is on by default only with `clientMetrics: "derive"`, so a
34
+ * config built without either option is unchanged.
20
35
  */
21
36
 
22
37
  import type { Declarable } from "@intentius/chant/declarable";
23
38
  import { GENAI_SEMCONV_PIN, type OTelComponent, type SchemaPin } from "./define";
24
39
  import { OtlpReceiver } from "./components/receivers";
25
- import { BatchProcessor, MemoryLimiterProcessor } from "./components/processors";
40
+ import { BatchProcessor, DeltaToCumulativeProcessor, MemoryLimiterProcessor } from "./components/processors";
26
41
  import { DebugExporter } from "./components/exporters";
27
42
  import { HealthCheckExtension } from "./components/extensions";
28
43
  import {
@@ -36,9 +51,13 @@ import {
36
51
  } from "./components/filtering";
37
52
  import {
38
53
  ForwardConnector,
54
+ SignalToMetricsConnector,
39
55
  SpanMetricsConnector,
40
56
  SumConnector,
41
57
  type ForwardConnectorConfig,
58
+ type SignalToMetricsAttribute,
59
+ type SignalToMetricsConnectorConfig,
60
+ type SignalToMetricsMetric,
42
61
  type SpanMetricsConnectorConfig,
43
62
  type SpanMetricsDimension,
44
63
  type SumConnectorConfig,
@@ -61,6 +80,11 @@ export const GENAI_ATTRIBUTES = Object.freeze({
61
80
  outputTokens: "gen_ai.usage.output_tokens",
62
81
  /** From the general conventions; GenAI spans set it on failure. */
63
82
  errorType: "error.type",
83
+ /** A metric attribute only: `input` or `output` on `gen_ai.client.token.usage`. */
84
+ tokenType: "gen_ai.token.type",
85
+ /** From the general conventions: the GenAI server's host and port. */
86
+ serverAddress: "server.address",
87
+ serverPort: "server.port",
64
88
  } as const);
65
89
 
66
90
  /**
@@ -88,6 +112,49 @@ export const GENAI_CONTENT_ATTRIBUTES: readonly string[] = Object.freeze([
88
112
  */
89
113
  export const GENAI_INDEXED_CONTENT_PATTERN = "^gen_ai\\.(prompt|completion)\\.[0-9]+\\..+$";
90
114
 
115
+ /**
116
+ * Attributes that take a new value per response, tool call, conversation,
117
+ * session or user. As a metric attribute, each one starts new time series
118
+ * for every value, so OTEL116 reports them (and the content keys above)
119
+ * wherever a connector splits a metric by one.
120
+ *
121
+ * At `GENAI_SEMCONV_PIN` (semantic-conventions v1.41.1) the conventions
122
+ * define `gen_ai.conversation.id`, `gen_ai.response.id` and
123
+ * `gen_ai.tool.call.id` in `model/gen-ai/registry.yaml`, `session.id` in
124
+ * `model/session/registry.yaml`, `user.id` in `model/user/registry.yaml` and
125
+ * `enduser.id` in `model/enduser/registry.yaml`. None of them is an attribute
126
+ * of any GenAI metric there (`model/gen-ai/metrics.yaml`).
127
+ * `gen_ai.request.previous_response.id` and `gen_ai.memory.record.id` are
128
+ * not at the pin: they come from the unreleased main of
129
+ * github.com/open-telemetry/semantic-conventions-genai, and are listed
130
+ * because instrumentation already sets them. Keys added with that pin
131
+ * (chant #3044) go here too.
132
+ */
133
+ export const GENAI_HIGH_CARDINALITY_ATTRIBUTES: readonly string[] = Object.freeze([
134
+ "gen_ai.conversation.id",
135
+ "gen_ai.response.id",
136
+ "gen_ai.tool.call.id",
137
+ "gen_ai.request.previous_response.id",
138
+ "gen_ai.memory.record.id",
139
+ "session.id",
140
+ "user.id",
141
+ "enduser.id",
142
+ ]);
143
+
144
+ const INDEXED_CONTENT = new RegExp(GENAI_INDEXED_CONTENT_PATTERN);
145
+
146
+ /**
147
+ * Why a key must not be a metric attribute: `"identifier"` for a key in
148
+ * `GENAI_HIGH_CARDINALITY_ATTRIBUTES`, `"content"` for a content key
149
+ * (`GENAI_CONTENT_ATTRIBUTES` or the indexed pattern), which is unbounded
150
+ * and sensitive too. Undefined for any other key.
151
+ */
152
+ export function genAiCardinalityRisk(key: string): "identifier" | "content" | undefined {
153
+ if (GENAI_HIGH_CARDINALITY_ATTRIBUTES.includes(key)) return "identifier";
154
+ if (GENAI_CONTENT_ATTRIBUTES.includes(key) || INDEXED_CONTENT.test(key)) return "content";
155
+ return undefined;
156
+ }
157
+
91
158
  /**
92
159
  * Deprecated content events. Their content is in the log record body, not in
93
160
  * attributes, so the preset deletes `content`, `message` and `tool_calls`
@@ -135,15 +202,82 @@ export const GENAI_DURATION_BUCKETS: readonly Duration[] = Object.freeze([
135
202
  "80s",
136
203
  ]);
137
204
 
205
+ /**
206
+ * Dimensions `providerDimensions: true` adds to `genai.calls` and
207
+ * `genai.duration`. The token sums keep the model alone (see
208
+ * `GENAI_TOKEN_DIMENSIONS`); `gen_ai.client.token.usage` splits tokens by both.
209
+ */
210
+ export const GENAI_PROVIDER_DIMENSIONS: readonly string[] = Object.freeze([
211
+ GENAI_ATTRIBUTES.providerName,
212
+ GENAI_ATTRIBUTES.responseModel,
213
+ ]);
214
+
215
+ // ── The conventions' client metrics, at GENAI_SEMCONV_PIN ────────────
216
+
217
+ /** The client metrics the conventions define that one span carries enough to derive. */
218
+ export const GENAI_CLIENT_METRIC_NAMES = Object.freeze({
219
+ operationDuration: "gen_ai.client.operation.duration",
220
+ tokenUsage: "gen_ai.client.token.usage",
221
+ } as const);
222
+
223
+ /** The values of `gen_ai.token.type` on `gen_ai.client.token.usage`. */
224
+ export const GENAI_TOKEN_TYPES = Object.freeze({ input: "input", output: "output" } as const);
225
+
226
+ /**
227
+ * The attributes the conventions give both client metrics: required
228
+ * (`gen_ai.operation.name`, `gen_ai.provider.name`), conditionally required
229
+ * (`gen_ai.request.model`, `error.type`) and recommended
230
+ * (`gen_ai.response.model`, `server.address`, `server.port`).
231
+ * `gen_ai.client.token.usage` adds `gen_ai.token.type`.
232
+ */
233
+ export const GENAI_CLIENT_METRIC_ATTRIBUTES: readonly string[] = Object.freeze([
234
+ GENAI_ATTRIBUTES.operationName,
235
+ GENAI_ATTRIBUTES.providerName,
236
+ GENAI_ATTRIBUTES.requestModel,
237
+ GENAI_ATTRIBUTES.responseModel,
238
+ GENAI_ATTRIBUTES.serverAddress,
239
+ GENAI_ATTRIBUTES.serverPort,
240
+ GENAI_ATTRIBUTES.errorType,
241
+ ]);
242
+
243
+ /** The bucket boundaries the conventions advise for `gen_ai.client.operation.duration`, in seconds. */
244
+ export const GENAI_CLIENT_DURATION_BUCKETS: readonly number[] = Object.freeze([
245
+ 0.01, 0.02, 0.04, 0.08, 0.16, 0.32, 0.64, 1.28, 2.56, 5.12, 10.24, 20.48, 40.96, 81.92,
246
+ ]);
247
+
248
+ /** The bucket boundaries the conventions advise for `gen_ai.client.token.usage`, in tokens. */
249
+ export const GENAI_CLIENT_TOKEN_BUCKETS: readonly number[] = Object.freeze([
250
+ 1, 4, 16, 64, 256, 1024, 4096, 16384, 65536, 262144, 1048576, 4194304, 16777216, 67108864,
251
+ ]);
252
+
253
+ /**
254
+ * Where the conventions' client metrics come from. `derive`: the collector
255
+ * builds them from spans. `passthrough`: the SDK already sends them, so the
256
+ * collector passes them on and derives nothing, since both would double
257
+ * every series.
258
+ */
259
+ export type GenAiClientMetricsSource = "derive" | "passthrough";
260
+
138
261
  /** One metric the preset emits, as the collector names it and as Prometheus exposes it. */
139
262
  export type GenAiMetric = CollectorMetric;
140
263
 
264
+ /** The conventions' client metrics, under their own names. */
265
+ export interface GenAiClientMetrics {
266
+ source: GenAiClientMetricsSource;
267
+ /** `gen_ai.client.operation.duration`, a histogram in seconds. Its `_count` is the operation count. */
268
+ operationDuration: GenAiMetric;
269
+ /** `gen_ai.client.token.usage`, a histogram in tokens; select `gen_ai_token_type` `input` or `output`. */
270
+ tokenUsage: GenAiMetric;
271
+ }
272
+
141
273
  export interface GenAiMetrics {
142
274
  /** Span count, with `status.code` = `STATUS_CODE_ERROR` for errors. */
143
275
  calls: GenAiMetric;
144
276
  duration: GenAiMetric;
145
277
  inputTokens: GenAiMetric;
146
278
  outputTokens: GenAiMetric;
279
+ /** The conventions' client metrics. Present only when `clientMetrics` is set. */
280
+ client?: GenAiClientMetrics;
147
281
  }
148
282
 
149
283
  export interface GenAiMetricsOptions {
@@ -151,6 +285,16 @@ export interface GenAiMetricsOptions {
151
285
  namespace?: string;
152
286
  /** Dimensions added to the span metrics beyond the GenAI ones. */
153
287
  dimensions?: SpanMetricsDimension[];
288
+ /** Add `gen_ai.provider.name` and `gen_ai.response.model` to `genai.calls` and `genai.duration`. Default false. */
289
+ providerDimensions?: boolean;
290
+ /**
291
+ * Also produce the conventions' client metrics, `gen_ai.client.operation.duration`
292
+ * and `gen_ai.client.token.usage`. `derive` builds them from spans;
293
+ * `passthrough` expects the SDK to send them. Either adds a `metrics`
294
+ * pipeline that passes the SDK's OTLP metrics through. Unset (the default)
295
+ * leaves the output as it was. The `genai.*` metrics are emitted either way.
296
+ */
297
+ clientMetrics?: GenAiClientMetricsSource;
154
298
  }
155
299
 
156
300
  const DEFAULT_NAMESPACE = "genai";
@@ -164,10 +308,10 @@ export function genAiMetrics(options: GenAiMetricsOptions = {}): GenAiMetrics {
164
308
  const ns = options.namespace ?? DEFAULT_NAMESPACE;
165
309
  const spanDims = [
166
310
  ...SPANMETRICS_DEFAULT_DIMENSIONS,
167
- ...GENAI_SPAN_METRIC_DIMENSIONS,
168
- ...(options.dimensions ?? []).map((d) => d.name),
311
+ ...spanMetricDimensions(options).map((d) => d.name),
169
312
  ];
170
313
  const tokenDims = [...GENAI_TOKEN_DIMENSIONS];
314
+ const source = clientMetricsSource(options);
171
315
  return {
172
316
  calls: { name: `${ns}.calls`, prometheus: prometheusMetricName(`${ns}.calls`, "sum"), type: "sum", dimensions: spanDims },
173
317
  duration: {
@@ -189,9 +333,49 @@ export function genAiMetrics(options: GenAiMetricsOptions = {}): GenAiMetrics {
189
333
  type: "sum",
190
334
  dimensions: tokenDims,
191
335
  },
336
+ ...(source ? { client: clientMetrics(source) } : {}),
337
+ };
338
+ }
339
+
340
+ function clientMetricsSource(options: GenAiMetricsOptions): GenAiClientMetricsSource | undefined {
341
+ const source = options.clientMetrics;
342
+ if (source === undefined) return undefined;
343
+ if (source !== "derive" && source !== "passthrough") {
344
+ throw new Error(`genAi: clientMetrics must be "derive" or "passthrough", got ${JSON.stringify(source)}`);
345
+ }
346
+ return source;
347
+ }
348
+
349
+ function clientMetrics(source: GenAiClientMetricsSource): GenAiClientMetrics {
350
+ const { operationDuration, tokenUsage } = GENAI_CLIENT_METRIC_NAMES;
351
+ return {
352
+ source,
353
+ operationDuration: {
354
+ name: operationDuration,
355
+ prometheus: prometheusMetricName(operationDuration, "histogram", "s"),
356
+ type: "histogram",
357
+ unit: "s",
358
+ dimensions: [...GENAI_CLIENT_METRIC_ATTRIBUTES],
359
+ },
360
+ tokenUsage: {
361
+ name: tokenUsage,
362
+ prometheus: prometheusMetricName(tokenUsage, "histogram", "{token}"),
363
+ type: "histogram",
364
+ unit: "{token}",
365
+ dimensions: [...GENAI_CLIENT_METRIC_ATTRIBUTES, GENAI_ATTRIBUTES.tokenType],
366
+ },
192
367
  };
193
368
  }
194
369
 
370
+ /** The spanmetrics dimensions beyond the connector's defaults, in config order. */
371
+ function spanMetricDimensions(options: GenAiMetricsOptions): SpanMetricsDimension[] {
372
+ return [
373
+ ...GENAI_SPAN_METRIC_DIMENSIONS.map((name) => ({ name })),
374
+ ...(options.providerDimensions ? GENAI_PROVIDER_DIMENSIONS.map((name) => ({ name })) : []),
375
+ ...(options.dimensions ?? []),
376
+ ];
377
+ }
378
+
195
379
  // ── Components ───────────────────────────────────────────────────────
196
380
 
197
381
  type Exporter = OTelComponent<"exporter", string, any>;
@@ -235,6 +419,19 @@ export interface GenAiComponents {
235
419
  spanMetrics: OTelComponent<"connector", "spanmetrics", SpanMetricsConnectorConfig>;
236
420
  /** Input and output token sums (`sum/genai_tokens`). */
237
421
  tokenUsage: OTelComponent<"connector", "sum", SumConnectorConfig>;
422
+ /**
423
+ * `gen_ai.client.operation.duration` and `gen_ai.client.token.usage` from
424
+ * spans (`signaltometrics/genai_client`). Present with `clientMetrics: "derive"`;
425
+ * wire it like `spanMetrics`.
426
+ */
427
+ clientMetrics?: OTelComponent<"connector", "signaltometrics", SignalToMetricsConnectorConfig>;
428
+ /**
429
+ * Drops the SDK's own copies of the derived client metrics from metrics
430
+ * passed through from the SDK (`filter/genai_sdk_client`), so each series
431
+ * is counted once. Present with `clientMetrics: "derive"`; put it on the
432
+ * pipeline that receives the SDK's OTLP metrics.
433
+ */
434
+ sdkClientMetricsFilter?: OTelComponent<"processor", "filter", FilterProcessorConfig>;
238
435
  metrics: GenAiMetrics;
239
436
  /** The semantic conventions the attribute keys follow. */
240
437
  semconv: SchemaPin;
@@ -315,7 +512,7 @@ export function genAiComponents(options: GenAiComponentsOptions = {}): GenAiComp
315
512
  const spanMetrics = new SpanMetricsConnector({
316
513
  name: "genai",
317
514
  namespace: options.namespace ?? DEFAULT_NAMESPACE,
318
- dimensions: [...GENAI_SPAN_METRIC_DIMENSIONS.map((name) => ({ name })), ...(options.dimensions ?? [])],
515
+ dimensions: spanMetricDimensions(options),
319
516
  histogram: { unit: "s", explicit: { buckets: [...(options.buckets ?? GENAI_DURATION_BUCKETS)] } },
320
517
  metrics_flush_interval: flush,
321
518
  });
@@ -342,6 +539,18 @@ export function genAiComponents(options: GenAiComponentsOptions = {}): GenAiComp
342
539
  },
343
540
  });
344
541
 
542
+ let clientMetricsConnector: GenAiComponents["clientMetrics"];
543
+ let sdkClientMetricsFilter: GenAiComponents["sdkClientMetricsFilter"];
544
+ if (metrics.client?.source === "derive") {
545
+ clientMetricsConnector = deriveClientMetrics(metrics.client, notAggregate);
546
+ const names = [metrics.client.operationDuration.name, metrics.client.tokenUsage.name];
547
+ sdkClientMetricsFilter = new FilterProcessor({
548
+ name: "genai_sdk_client",
549
+ error_mode: "ignore",
550
+ metrics: { metric: names.map((n) => `name == ${quote(n)}`) },
551
+ });
552
+ }
553
+
345
554
  const processors: Processor[] = [];
346
555
  if (contentRemoval) processors.push(contentRemoval);
347
556
  if (redaction) processors.push(redaction);
@@ -354,17 +563,90 @@ export function genAiComponents(options: GenAiComponentsOptions = {}): GenAiComp
354
563
  genAiSpans,
355
564
  spanMetrics,
356
565
  tokenUsage,
566
+ ...(clientMetricsConnector ? { clientMetrics: clientMetricsConnector } : {}),
567
+ ...(sdkClientMetricsFilter ? { sdkClientMetricsFilter } : {}),
357
568
  metrics,
358
569
  semconv: GENAI_SEMCONV_PIN,
359
570
  };
360
571
  }
361
572
 
573
+ /**
574
+ * The `signaltometrics` connector for the conventions' client metrics. Every
575
+ * GenAI span records its duration. Token usage is one histogram fed by two
576
+ * entries, input and output, each setting `gen_ai.token.type` through the
577
+ * attribute's `default_value` (spans don't carry that key); the connector
578
+ * merges entries with the same name, unit and description into one metric.
579
+ * Token entries skip the in-process agent and workflow spans the `sum`
580
+ * connector skips, and spans whose count is not a number, which would
581
+ * otherwise fail the batch.
582
+ */
583
+ function deriveClientMetrics(client: GenAiClientMetrics, notAggregate: string) {
584
+ const [operation, ...rest] = GENAI_CLIENT_METRIC_ATTRIBUTES;
585
+ // Spans always have the operation (filter/genai_spans keeps no other); the
586
+ // rest are optional, so a span that lacks one, such as a tool call without
587
+ // a provider, is still counted.
588
+ const attributes: SignalToMetricsAttribute[] = [{ key: operation! }, ...rest.map((key) => ({ key, optional: true }))];
589
+ const duration: SignalToMetricsMetric = {
590
+ name: client.operationDuration.name,
591
+ description: "GenAI operation duration.",
592
+ unit: client.operationDuration.unit!,
593
+ attributes,
594
+ histogram: {
595
+ buckets: [...GENAI_CLIENT_DURATION_BUCKETS],
596
+ value: "Double(Microseconds(end_time - start_time)) / 1000000.0",
597
+ },
598
+ };
599
+ const tokens = (tokenType: string, source: string): SignalToMetricsMetric => {
600
+ const attr = `attributes[${quote(source)}]`;
601
+ return {
602
+ name: client.tokenUsage.name,
603
+ description: "Number of input and output tokens used.",
604
+ unit: client.tokenUsage.unit!,
605
+ attributes: [...attributes, { key: GENAI_ATTRIBUTES.tokenType, default_value: tokenType }],
606
+ conditions: [`(IsInt(${attr}) or IsDouble(${attr})) and ${notAggregate}`],
607
+ histogram: { buckets: [...GENAI_CLIENT_TOKEN_BUCKETS], value: attr },
608
+ };
609
+ };
610
+ return new SignalToMetricsConnector({
611
+ name: "genai_client",
612
+ spans: [
613
+ duration,
614
+ tokens(GENAI_TOKEN_TYPES.input, GENAI_ATTRIBUTES.inputTokens),
615
+ tokens(GENAI_TOKEN_TYPES.output, GENAI_ATTRIBUTES.outputTokens),
616
+ ],
617
+ });
618
+ }
619
+
362
620
  // ── The whole collector ──────────────────────────────────────────────
363
621
 
622
+ /**
623
+ * Metric exporters that take delta sums and histograms as they are: the
624
+ * `prometheus` exporter accumulates them itself, and `debug` prints them.
625
+ * Any other exporter gets cumulative data under `deltaToCumulative: "auto"`.
626
+ * `prometheusremotewrite`, for one, drops non-cumulative monotonic sums,
627
+ * histograms and summaries (its README at collector-contrib v0.130.0).
628
+ */
629
+ export const DELTA_READY_EXPORTERS: readonly string[] = Object.freeze(["prometheus", "debug"]);
630
+
631
+ /** Whether `deltaToCumulative` puts a `deltatocumulative` processor on `metrics/genai` for these exporters. */
632
+ export function genAiNeedsDeltaToCumulative(
633
+ setting: GenAiPipelineOptions["deltaToCumulative"],
634
+ metricExporters: readonly Exporter[],
635
+ ): boolean {
636
+ if (setting === true) return true;
637
+ if (setting === "auto") return metricExporters.some((e) => !DELTA_READY_EXPORTERS.includes(e.componentType));
638
+ if (setting === undefined || setting === false) return false;
639
+ throw new Error(`genAiPipeline: deltaToCumulative must be true, false or "auto", got ${JSON.stringify(setting)}`);
640
+ }
641
+
364
642
  export interface GenAiPipelineOptions extends GenAiComponentsOptions {
365
643
  /** Where traces go. Default: one `debug` exporter at `basic` verbosity. */
366
644
  traceExporters?: Exporter[];
367
- /** Where the span and token metrics go. Default: the same `debug` exporter. */
645
+ /**
646
+ * Where the span and token metrics go, and with `clientMetrics` the SDK's
647
+ * metrics too. Default: the same `debug` exporter. See `deltaToCumulative`
648
+ * for exporters that need cumulative data.
649
+ */
368
650
  metricExporters?: Exporter[];
369
651
  /** Where logs (and events sent as logs) go. Default: the same `debug` exporter. */
370
652
  logExporters?: Exporter[];
@@ -378,6 +660,21 @@ export interface GenAiPipelineOptions extends GenAiComponentsOptions {
378
660
  sampling?: Processor[];
379
661
  /** Serve `health_check` on 0.0.0.0:13133. Default: true. */
380
662
  healthCheck?: boolean;
663
+ /**
664
+ * Put a `deltatocumulative/genai` processor in front of `batch` on
665
+ * `metrics/genai`, so the delta token sums of the `sum` connector, and with
666
+ * `clientMetrics: "derive"` the delta histograms of `signaltometrics`, reach
667
+ * exporters as cumulative data. The span metrics are cumulative already and
668
+ * pass through unchanged. `"auto"`: when a metric exporter is not in
669
+ * `DELTA_READY_EXPORTERS` (`prometheus`, `debug`), such as
670
+ * `prometheusremotewrite` or `otlp`. `true`: always. `false`: never, such
671
+ * as for an OTLP backend that wants deltas. Unset: `"auto"` with
672
+ * `clientMetrics: "derive"`, otherwise `false`, which is the output from
673
+ * before the option existed. The processor keeps running totals in memory,
674
+ * so a collector behind a load balancer needs each stream to reach one
675
+ * replica.
676
+ */
677
+ deltaToCumulative?: boolean | "auto";
381
678
  }
382
679
 
383
680
  /**
@@ -393,6 +690,12 @@ export function genAiPipeline(options: GenAiPipelineOptions = {}): Declarable[]
393
690
  const { traceExporters = [debug], metricExporters = [debug], logExporters = [debug], logs = true, sampling = [], healthCheck = true } =
394
691
  options;
395
692
  const parts = genAiComponents(options);
693
+ const client = parts.metrics.client;
694
+ // derive mode is opt-in and newer than the option, so it can default to "auto".
695
+ const cumulativeSetting = options.deltaToCumulative ?? (client?.source === "derive" ? "auto" : false);
696
+ const cumulative = genAiNeedsDeltaToCumulative(cumulativeSetting, metricExporters)
697
+ ? [new DeltaToCumulativeProcessor({ name: "genai" })]
698
+ : [];
396
699
 
397
700
  const otlp = new OtlpReceiver({
398
701
  protocols: {
@@ -441,16 +744,28 @@ export function genAiPipeline(options: GenAiPipelineOptions = {}): Declarable[]
441
744
  name: "genai",
442
745
  receivers: [parts.forward],
443
746
  processors: [parts.genAiSpans],
444
- exporters: [parts.spanMetrics, parts.tokenUsage],
747
+ exporters: [parts.spanMetrics, parts.tokenUsage, ...(parts.clientMetrics ? [parts.clientMetrics] : [])],
445
748
  }),
446
749
  new Pipeline({
447
750
  signal: "metrics",
448
751
  name: "genai",
449
- receivers: [parts.spanMetrics, parts.tokenUsage],
450
- processors: [batch],
752
+ receivers: [parts.spanMetrics, parts.tokenUsage, ...(parts.clientMetrics ? [parts.clientMetrics] : [])],
753
+ processors: [...cumulative, batch],
451
754
  exporters: metricExporters,
452
755
  }),
453
756
  );
757
+ if (client) {
758
+ // The SDK's own metrics: the client metrics in passthrough, and the
759
+ // ones no span carries (time to first chunk, gen_ai.server.*) either way.
760
+ entities.push(
761
+ new Pipeline({
762
+ signal: "metrics",
763
+ receivers: [otlp],
764
+ processors: [memoryLimiter, ...(parts.sdkClientMetricsFilter ? [parts.sdkClientMetricsFilter] : []), batch],
765
+ exporters: metricExporters,
766
+ }),
767
+ );
768
+ }
454
769
  if (logs) {
455
770
  entities.push(
456
771
  new Pipeline({
@@ -19,6 +19,11 @@
19
19
  "kind": "resource",
20
20
  "lexicon": "otel"
21
21
  },
22
+ "DeltaToCumulativeProcessor": {
23
+ "resourceType": "OTel::Processor::deltatocumulative",
24
+ "kind": "resource",
25
+ "lexicon": "otel"
26
+ },
22
27
  "FileLogReceiver": {
23
28
  "resourceType": "OTel::Receiver::filelog",
24
29
  "kind": "resource",
@@ -59,6 +64,11 @@
59
64
  "kind": "resource",
60
65
  "lexicon": "otel"
61
66
  },
67
+ "K8sLeaderElectorExtension": {
68
+ "resourceType": "OTel::Extension::k8s_leader_elector",
69
+ "kind": "resource",
70
+ "lexicon": "otel"
71
+ },
62
72
  "KubeletStatsReceiver": {
63
73
  "resourceType": "OTel::Receiver::kubeletstats",
64
74
  "kind": "resource",
@@ -144,6 +154,11 @@
144
154
  "kind": "resource",
145
155
  "lexicon": "otel"
146
156
  },
157
+ "SignalToMetricsConnector": {
158
+ "resourceType": "OTel::Connector::signaltometrics",
159
+ "kind": "resource",
160
+ "lexicon": "otel"
161
+ },
147
162
  "SpanMetricsConnector": {
148
163
  "resourceType": "OTel::Connector::spanmetrics",
149
164
  "kind": "resource",
@@ -155,7 +155,7 @@ function otelcolValidate(yaml: string): { ok: boolean; output: string } {
155
155
  describe("YAML -> TypeScript -> YAML", () => {
156
156
  test("the otel examples' built output", async () => {
157
157
  const outputs = await exampleOutputs();
158
- expect(outputs.map(([n]) => n)).toEqual(["custom-component", "genai-agent", "getting-started", "k8s-node-agent"]);
158
+ expect(outputs.map(([n]) => n)).toEqual(["custom-component", "genai-agent", "getting-started", "k8s-node-agent", "tail-sampling-gateway"]);
159
159
  for (const [name, yaml] of outputs) {
160
160
  const out = await expectRoundTrip(yaml);
161
161
  // The `# chant:` header comes back too: the custom component's pin, the semconv line.
@@ -197,6 +197,25 @@ describe("YAML -> TypeScript -> YAML", () => {
197
197
  }
198
198
  });
199
199
 
200
+ test("a signaltometrics connector imports to the typed class", async () => {
201
+ const out = await expectRoundTrip(read("signaltometrics.yaml"));
202
+ expect(out.source).toContain("new SignalToMetricsConnector(");
203
+ expect(out.source).not.toContain("defineComponent");
204
+ // The connector is one entity, on both sides of the join.
205
+ expect(out.source).toContain("receivers: [signaltometricsGenai]");
206
+ // A constant OTTL value stays a string.
207
+ expect(load(out.yaml)).toMatchObject({ connectors: { "signaltometrics/genai": { logs: [{ sum: { value: "1" } }] } } });
208
+ expect(out.warnings).toEqual([]);
209
+ });
210
+
211
+ test("a k8s_leader_elector extension imports to the typed class", async () => {
212
+ const out = await expectRoundTrip(read("k8s-leader-elector.yaml"));
213
+ expect(out.source).toContain("new K8sLeaderElectorExtension(");
214
+ expect(out.source).not.toContain("defineComponent");
215
+ expect(out.source).toContain('lease_name: "otel-k8s-cluster"');
216
+ expect(out.warnings).toEqual([]);
217
+ });
218
+
200
219
  test("the collector configs of examples/agent-observability", async () => {
201
220
  for (const file of ["agent-observability-agent.yaml", "agent-observability-gateway.yaml"]) {
202
221
  await expectRoundTrip(read(file));
@@ -214,6 +233,7 @@ describe("YAML -> TypeScript -> YAML", () => {
214
233
  // Not the agent-observability agent: its k8s resolver needs a cluster to build (that example's own tests swap it for dns).
215
234
  const yamls: Array<[string, string]> = [
216
235
  ["gateway.yaml", read("gateway.yaml")],
236
+ ["signaltometrics.yaml", read("signaltometrics.yaml")],
217
237
  ["agent-observability-gateway.yaml", read("agent-observability-gateway.yaml")],
218
238
  ["genAiPipeline()", collectorYaml(genAiPipeline())],
219
239
  ...UPSTREAM_BUILTIN_ONLY.map((f): [string, string] => [f, read("upstream", f)]),
@@ -82,6 +82,7 @@ export function everyBuiltin(): Declarable[] {
82
82
  node_conditions_to_report: ["Ready", "MemoryPressure"],
83
83
  allocatable_types_to_report: ["cpu", "memory"],
84
84
  metrics: { "k8s.pod.phase": { enabled: false } },
85
+ k8s_leader_elector: "k8s_leader_elector/cluster",
85
86
  });
86
87
  const kubelet = new c.KubeletStatsReceiver({
87
88
  auth_type: "serviceAccount",
@@ -154,6 +155,7 @@ export function everyBuiltin(): Declarable[] {
154
155
  ],
155
156
  });
156
157
  const probabilistic = new c.ProbabilisticSamplerProcessor({ sampling_percentage: 12.5, mode: "proportional", sampling_precision: 4 });
158
+ const deltaToCumulative = new c.DeltaToCumulativeProcessor({ max_stale: "10m", max_streams: 50000 });
157
159
 
158
160
  const otlpOut = new c.OtlpExporter({
159
161
  name: "backend",
@@ -213,10 +215,32 @@ export function everyBuiltin(): Declarable[] {
213
215
  const sum = new c.SumConnector({
214
216
  spans: { "span.bytes": { source_attribute: "bytes", description: "Bytes by route", conditions: ['attributes["bytes"] != nil'] } },
215
217
  });
218
+ const signalToMetrics = new c.SignalToMetricsConnector({
219
+ spans: [
220
+ {
221
+ name: "span.duration",
222
+ unit: "ms",
223
+ attributes: [{ key: "http.route", default_value: "none" }, { key: "error.type", optional: true }],
224
+ include_resource_attributes: [{ key: "service.name" }],
225
+ conditions: ["kind == SPAN_KIND_SERVER"],
226
+ histogram: { buckets: [5, 50, 500], value: "Milliseconds(end_time - start_time)" },
227
+ },
228
+ ],
229
+ logs: [{ name: "logrecord.count", sum: { value: "1" } }],
230
+ });
216
231
 
217
232
  const health = new c.HealthCheckExtension({ endpoint: "0.0.0.0:13133", path: "/health", response_body: { healthy: "ok" } });
218
233
  const pprof = new c.PprofExtension({ endpoint: "localhost:1777", block_profile_fraction: 3 });
219
234
  const zpages = new c.ZPagesExtension({ endpoint: "localhost:55679" });
235
+ const leaderElector = new c.K8sLeaderElectorExtension({
236
+ name: "cluster",
237
+ auth_type: "serviceAccount",
238
+ lease_name: "otel-k8s-cluster",
239
+ lease_namespace: "observability",
240
+ lease_duration: "20s",
241
+ renew_deadline: "15s",
242
+ retry_period: "3s",
243
+ });
220
244
 
221
245
  return [
222
246
  otlp,
@@ -236,6 +260,7 @@ export function everyBuiltin(): Declarable[] {
236
260
  redaction,
237
261
  tailSampling,
238
262
  probabilistic,
263
+ deltaToCumulative,
239
264
  otlpOut,
240
265
  otlphttp,
241
266
  debug,
@@ -248,26 +273,33 @@ export function everyBuiltin(): Declarable[] {
248
273
  forward,
249
274
  count,
250
275
  sum,
276
+ signalToMetrics,
251
277
  health,
252
278
  pprof,
253
279
  zpages,
280
+ leaderElector,
254
281
  new Pipeline({
255
282
  signal: "traces",
256
283
  receivers: [otlp],
257
284
  processors: [memoryLimiter, k8sattributes, resourcedetection, resource, attributes, filter, transform, redaction, probabilistic],
258
- exporters: [spanmetrics, servicegraph, count, sum, forward, routing, loadbalancing],
285
+ exporters: [spanmetrics, servicegraph, count, sum, signalToMetrics, forward, routing, loadbalancing],
259
286
  }),
260
287
  new Pipeline({ signal: "traces", name: "acme", receivers: [routing], processors: [tailSampling, batch], exporters: [otlpOut] }),
261
288
  new Pipeline({ signal: "traces", name: "rest", receivers: [routing, forward], processors: [batch], exporters: [googlecloud, debug] }),
262
289
  new Pipeline({
263
290
  signal: "metrics",
264
- receivers: [otlp, prometheusIn, hostmetrics, cluster, kubelet, spanmetrics, servicegraph, count, sum],
265
- processors: [memoryLimiter, filter, batch],
291
+ receivers: [otlp, prometheusIn, hostmetrics, cluster, kubelet, spanmetrics, servicegraph, count, sum, signalToMetrics],
292
+ processors: [memoryLimiter, filter, deltaToCumulative, batch],
266
293
  exporters: [prometheusOut, otlphttp],
267
294
  }),
268
- new Pipeline({ signal: "logs", receivers: [otlp, filelog], processors: [memoryLimiter, redaction, batch], exporters: [otlphttp, debug] }),
295
+ new Pipeline({
296
+ signal: "logs",
297
+ receivers: [otlp, filelog],
298
+ processors: [memoryLimiter, redaction, batch],
299
+ exporters: [otlphttp, debug, signalToMetrics],
300
+ }),
269
301
  new Service({
270
- extensions: [health, zpages, pprof],
302
+ extensions: [health, zpages, pprof, leaderElector],
271
303
  telemetry: { logs: { level: "warn", encoding: "json" }, metrics: { level: "normal" }, resource: { "service.name": "gw" } },
272
304
  }),
273
305
  ];
@@ -0,0 +1,35 @@
1
+ # A k8s_cluster receiver behind a k8s_leader_elector extension
2
+ # (collector-contrib v0.130.0), for a gateway with more than one replica:
3
+ # only the replica holding the Lease collects.
4
+
5
+ receivers:
6
+ k8s_cluster:
7
+ auth_type: serviceAccount
8
+ collection_interval: 30s
9
+ k8s_leader_elector: k8s_leader_elector/cluster
10
+ otlp:
11
+ protocols:
12
+ grpc:
13
+ endpoint: 0.0.0.0:4317
14
+ processors:
15
+ batch: {}
16
+ exporters:
17
+ otlp:
18
+ endpoint: backend.observability.svc:4317
19
+ extensions:
20
+ health_check:
21
+ endpoint: 0.0.0.0:13133
22
+ k8s_leader_elector/cluster:
23
+ auth_type: serviceAccount
24
+ lease_name: otel-k8s-cluster
25
+ lease_namespace: observability
26
+ lease_duration: 20s
27
+ renew_deadline: 15s
28
+ retry_period: 3s
29
+ service:
30
+ extensions: [health_check, k8s_leader_elector/cluster]
31
+ pipelines:
32
+ metrics:
33
+ receivers: [otlp, k8s_cluster]
34
+ processors: [batch]
35
+ exporters: [otlp]