@intentius/chant-lexicon-prometheus 0.100.0 → 0.101.0
This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
- package/README.md +4 -1
- package/dist/codegen/docs.d.ts.map +1 -1
- package/dist/composites/catalog.d.ts.map +1 -1
- package/dist/composites/genai.d.ts +224 -0
- package/dist/composites/genai.d.ts.map +1 -0
- package/dist/composites/index.d.ts +2 -0
- package/dist/composites/index.d.ts.map +1 -1
- package/dist/index.d.ts +1 -1
- package/dist/index.d.ts.map +1 -1
- package/dist/init-templates.d.ts +11 -2
- package/dist/init-templates.d.ts.map +1 -1
- package/dist/integrity.json +3 -3
- package/dist/manifest.json +1 -1
- package/dist/plugin.d.ts.map +1 -1
- package/dist/rule-eval.d.ts +10 -4
- package/dist/rule-eval.d.ts.map +1 -1
- package/dist/skills/chant-prometheus.md +22 -0
- package/package.json +3 -2
- package/src/codegen/docs.ts +5 -0
- package/src/composites/catalog.test.ts +1 -1
- package/src/composites/catalog.ts +70 -0
- package/src/composites/genai.test.ts +307 -0
- package/src/composites/genai.ts +678 -0
- package/src/composites/index.ts +15 -0
- package/src/composites/slo-burn.test.ts +25 -1
- package/src/index.ts +14 -0
- package/src/init-templates.test.ts +37 -11
- package/src/init-templates.ts +75 -6
- package/src/plugin.ts +6 -2
- package/src/rule-eval.ts +72 -7
- package/src/skills/chant-prometheus.md +22 -0
- package/src/typecheck.test.ts +88 -0
|
@@ -0,0 +1,678 @@
|
|
|
1
|
+
/**
|
|
2
|
+
* `GenAiRules`: recording rules and opt-in alerts for the otel lexicon's
|
|
3
|
+
* GenAI collector preset, and spend from a price table the project declares.
|
|
4
|
+
*
|
|
5
|
+
* Every metric and label name comes from the `GenAiMetrics` the collector was
|
|
6
|
+
* built with (`genAiMetrics(options)` or `genAiComponents(options).metrics`)
|
|
7
|
+
* and from the otel lexicon's attribute keys, so a collector with another
|
|
8
|
+
* namespace, or with the conventions' client metrics switched on, moves every
|
|
9
|
+
* expression here with it.
|
|
10
|
+
*
|
|
11
|
+
* Two sources describe model calls:
|
|
12
|
+
*
|
|
13
|
+
* - `client`: the conventions' `gen_ai.client.operation.duration` and
|
|
14
|
+
* `gen_ai.client.token.usage` (#3041). Requests are the duration
|
|
15
|
+
* histogram's `_count`, errors the ones with an `error.type`, tokens the
|
|
16
|
+
* token histogram's `_sum` split by `gen_ai.token.type`.
|
|
17
|
+
* - `spans`: the preset's own `genai.calls`, `genai.duration` and token sums.
|
|
18
|
+
* Errors are spans with `status.code` `STATUS_CODE_ERROR`. The token sums
|
|
19
|
+
* carry the model only, so cost series take the provider from the price
|
|
20
|
+
* table.
|
|
21
|
+
*
|
|
22
|
+
* The default is `client` when the metrics have it. Per-tool rules always
|
|
23
|
+
* read the span metrics: the conventions' client metrics carry no tool name.
|
|
24
|
+
*
|
|
25
|
+
* Everything is in one group, recording rules first, because Prometheus
|
|
26
|
+
* evaluates a group's rules in order and the later rules and the alerts read
|
|
27
|
+
* the series the earlier ones record in the same evaluation.
|
|
28
|
+
*
|
|
29
|
+
* Cost is a rate in the price's currency per second, one series per priced
|
|
30
|
+
* model and token type, labelled with the currency. A model the table does
|
|
31
|
+
* not price gets no cost series, never a cost of zero. The lexicon ships no
|
|
32
|
+
* prices. `currency` and `source` are required on every price, the same two
|
|
33
|
+
* fields a workspace run's cost record carries (#3033), so the two can be
|
|
34
|
+
* compared; chant converts no currency.
|
|
35
|
+
*/
|
|
36
|
+
|
|
37
|
+
import { Composite, type CompositeInstance } from "@intentius/chant/composite";
|
|
38
|
+
import { GENAI_ATTRIBUTES, GENAI_TOKEN_TYPES, type GenAiMetric, type GenAiMetrics } from "@intentius/chant-lexicon-otel/genai";
|
|
39
|
+
import { prometheusLabel, SPAN_STATUS_ERROR } from "@intentius/chant-lexicon-otel/metric-names";
|
|
40
|
+
import { RuleGroup, type AlertingRule, type RecordingRule, type RuleGroupEntity } from "../rules";
|
|
41
|
+
import type { LabelSet } from "../model";
|
|
42
|
+
import { durationMs, isValidDuration } from "../duration";
|
|
43
|
+
|
|
44
|
+
/** One model's price, as the provider publishes it, per million tokens. */
|
|
45
|
+
export interface GenAiPrice {
|
|
46
|
+
/** The `gen_ai.provider.name` value, e.g. `anthropic` or `openai`. */
|
|
47
|
+
provider: string;
|
|
48
|
+
/** The `gen_ai.request.model` value the price applies to, exactly as spans report it. */
|
|
49
|
+
model: string;
|
|
50
|
+
/** Price of one million input tokens. */
|
|
51
|
+
inputPerMTok: number;
|
|
52
|
+
/** Price of one million output tokens. */
|
|
53
|
+
outputPerMTok: number;
|
|
54
|
+
/** The currency the prices are in, e.g. `USD`. Required: chant converts nothing. */
|
|
55
|
+
currency: string;
|
|
56
|
+
/** Where the prices come from, e.g. the provider's pricing page URL. Required. */
|
|
57
|
+
source: string;
|
|
58
|
+
/** The date the prices were read, `YYYY-MM-DD`. */
|
|
59
|
+
asOf?: string;
|
|
60
|
+
}
|
|
61
|
+
|
|
62
|
+
/** The fields every alert takes. */
|
|
63
|
+
export interface GenAiAlertOptions {
|
|
64
|
+
/** How long the condition must hold (default `10m`). */
|
|
65
|
+
for?: string;
|
|
66
|
+
/** The `severity` label, which Alertmanager routes on (default `warning`). */
|
|
67
|
+
severity?: string;
|
|
68
|
+
/** More labels on the alert. */
|
|
69
|
+
labels?: LabelSet;
|
|
70
|
+
/** More annotations on the alert, e.g. `runbook_url`. */
|
|
71
|
+
annotations?: LabelSet;
|
|
72
|
+
}
|
|
73
|
+
|
|
74
|
+
export interface GenAiRatioAlert extends GenAiAlertOptions {
|
|
75
|
+
/** The error ratio the alert fires above, between 0 and 1. */
|
|
76
|
+
threshold?: number;
|
|
77
|
+
}
|
|
78
|
+
|
|
79
|
+
export interface GenAiLatencyAlert extends GenAiAlertOptions {
|
|
80
|
+
/** The p95 operation latency, in seconds, the alert fires above (default 30). */
|
|
81
|
+
thresholdSeconds?: number;
|
|
82
|
+
}
|
|
83
|
+
|
|
84
|
+
/** A spending limit over the last hour or day, in one currency. */
|
|
85
|
+
export interface GenAiBudget extends GenAiAlertOptions {
|
|
86
|
+
/** The most the window may cost, e.g. `50`. */
|
|
87
|
+
amount: number;
|
|
88
|
+
/** The currency of `amount`; it must be the currency of at least one price. */
|
|
89
|
+
currency: string;
|
|
90
|
+
/** `hour` (spend over the last hour) or `day` (the last 24 hours). */
|
|
91
|
+
per: "hour" | "day";
|
|
92
|
+
}
|
|
93
|
+
|
|
94
|
+
/** Opt-in alerts. None is built unless it is set here. */
|
|
95
|
+
export interface GenAiAlerting {
|
|
96
|
+
/** Error ratio per provider, model and operation (default threshold 0.05). */
|
|
97
|
+
errorRatio?: true | GenAiRatioAlert;
|
|
98
|
+
/** p95 operation latency per provider, model and operation (default 30s). */
|
|
99
|
+
latency?: true | GenAiLatencyAlert;
|
|
100
|
+
/** Error ratio per tool, from the span metrics (default threshold 0.1). */
|
|
101
|
+
toolErrorRatio?: true | GenAiRatioAlert;
|
|
102
|
+
/** Spend over a budget per hour or per day, from the cost rules. Needs `prices`. */
|
|
103
|
+
budgets?: GenAiBudget[];
|
|
104
|
+
}
|
|
105
|
+
|
|
106
|
+
export interface GenAiRulesProps {
|
|
107
|
+
/** The preset's metrics: `genAiMetrics(options)` with the options the collector was built with, or `genAiComponents(options)`. */
|
|
108
|
+
genAi: GenAiMetrics | { metrics: GenAiMetrics };
|
|
109
|
+
/** Which metrics the model rules read (default `client` when the metrics include the conventions' client metrics, else `spans`). */
|
|
110
|
+
source?: "client" | "spans";
|
|
111
|
+
/** Prices per provider and model. A model missing here gets no cost series. */
|
|
112
|
+
prices?: GenAiPrice[];
|
|
113
|
+
/** Opt-in alerts (default: none). */
|
|
114
|
+
alerts?: GenAiAlerting;
|
|
115
|
+
/** The rule group's name (default `genai`). */
|
|
116
|
+
name?: string;
|
|
117
|
+
/** The first part of every recorded series name (default `gen_ai`). */
|
|
118
|
+
prefix?: string;
|
|
119
|
+
/** The range every `rate` reads (default `5m`). */
|
|
120
|
+
rateWindow?: string;
|
|
121
|
+
/** More Prometheus labels every rule keeps, e.g. `job` or `service_name`. */
|
|
122
|
+
groupBy?: string[];
|
|
123
|
+
/** Labels added to every rule, e.g. `team`. */
|
|
124
|
+
labels?: LabelSet;
|
|
125
|
+
/** Evaluation interval of the group (default: Prometheus's `evaluation_interval`). */
|
|
126
|
+
interval?: string;
|
|
127
|
+
}
|
|
128
|
+
|
|
129
|
+
export type GenAiRulesMembers = {
|
|
130
|
+
/** The recording rules, then the alerts. */
|
|
131
|
+
rules: RuleGroupEntity;
|
|
132
|
+
};
|
|
133
|
+
|
|
134
|
+
/** What `GenAiRules(...)` returns: its rule group, as `rules`. */
|
|
135
|
+
export type GenAiRulesInstance = CompositeInstance<GenAiRulesMembers> & GenAiRulesMembers;
|
|
136
|
+
|
|
137
|
+
/** A recorded latency series and the quantiles it holds, one per `quantile` label value. */
|
|
138
|
+
export interface GenAiQuantileSeries {
|
|
139
|
+
record: string;
|
|
140
|
+
/** The values of the `quantile` label, e.g. `0.95`. */
|
|
141
|
+
quantiles: string[];
|
|
142
|
+
}
|
|
143
|
+
|
|
144
|
+
/** One alert as built. */
|
|
145
|
+
export interface GenAiAlertInfo {
|
|
146
|
+
alert: string;
|
|
147
|
+
kind: "errorRatio" | "latency" | "toolErrorRatio" | "budget";
|
|
148
|
+
severity: string;
|
|
149
|
+
/** The value the alert fires above: a ratio, seconds, or an amount of money. */
|
|
150
|
+
threshold: number;
|
|
151
|
+
/** For a budget: its currency and window. */
|
|
152
|
+
currency?: string;
|
|
153
|
+
per?: "hour" | "day";
|
|
154
|
+
}
|
|
155
|
+
|
|
156
|
+
/** The series a `GenAiRules` records, read by dashboards instead of repeating names. */
|
|
157
|
+
export interface GenAiRuleMetrics {
|
|
158
|
+
/** The rule group's name. */
|
|
159
|
+
group: string;
|
|
160
|
+
/** Which metrics the model rules read. */
|
|
161
|
+
source: "client" | "spans";
|
|
162
|
+
rateWindow: string;
|
|
163
|
+
/** Prometheus label names the series are split by. */
|
|
164
|
+
labels: {
|
|
165
|
+
/** Absent when the source metrics have no provider (span metrics without `providerDimensions`). */
|
|
166
|
+
provider?: string;
|
|
167
|
+
model: string;
|
|
168
|
+
operation: string;
|
|
169
|
+
errorType: string;
|
|
170
|
+
tokenType: string;
|
|
171
|
+
/** Absent when the span metrics have no tool dimension. */
|
|
172
|
+
tool?: string;
|
|
173
|
+
currency: string;
|
|
174
|
+
quantile: string;
|
|
175
|
+
};
|
|
176
|
+
/** The labels the model series (requests, errors, latency) are split by, `groupBy` last. */
|
|
177
|
+
modelLabels: string[];
|
|
178
|
+
/** The labels token and cost series are split by, besides the token type (and currency on cost). */
|
|
179
|
+
tokenLabels: string[];
|
|
180
|
+
/** Requests per second. */
|
|
181
|
+
requests: string;
|
|
182
|
+
/** Errors per second, also by `error.type`. */
|
|
183
|
+
errors: string;
|
|
184
|
+
/** Errors over requests, 0 when there are none. */
|
|
185
|
+
errorRatio: string;
|
|
186
|
+
/** Errors of each `error.type` over all requests. */
|
|
187
|
+
errorRatioByType: string;
|
|
188
|
+
/** Operation latency in seconds, at p50, p95 and p99. */
|
|
189
|
+
latency: GenAiQuantileSeries;
|
|
190
|
+
/** Tokens per second, by `gen_ai.token.type` (`input`, `output`). */
|
|
191
|
+
tokens: string;
|
|
192
|
+
/** Spend per second in each price's currency, by token type. Absent without prices. */
|
|
193
|
+
cost?: string;
|
|
194
|
+
/** Per-tool series, from the span metrics. Absent when they carry no tool name. */
|
|
195
|
+
tool?: {
|
|
196
|
+
calls: string;
|
|
197
|
+
errors: string;
|
|
198
|
+
errorRatio: string;
|
|
199
|
+
latency: GenAiQuantileSeries;
|
|
200
|
+
};
|
|
201
|
+
/** The price table as declared, without the prices. */
|
|
202
|
+
prices: Array<{ provider: string; model: string; currency: string; source: string; asOf?: string }>;
|
|
203
|
+
/** Every currency in the price table. */
|
|
204
|
+
currencies: string[];
|
|
205
|
+
/** The alerts built, in rule order. Empty unless asked for. */
|
|
206
|
+
alerts: GenAiAlertInfo[];
|
|
207
|
+
}
|
|
208
|
+
|
|
209
|
+
const GENAI_RULES_METRICS = Symbol.for("chant.prometheus.genai");
|
|
210
|
+
const PREFIX = /^[A-Za-z_][A-Za-z0-9_]*$/;
|
|
211
|
+
const LABEL = /^[A-Za-z_][A-Za-z0-9_]*$/;
|
|
212
|
+
const GROUP_NAME = /^[A-Za-z0-9][A-Za-z0-9_.-]*$/;
|
|
213
|
+
const DATE = /^\d{4}-\d{2}-\d{2}$/;
|
|
214
|
+
const QUANTILES = ["0.5", "0.95", "0.99"] as const;
|
|
215
|
+
const ALERT_QUANTILE = "0.95";
|
|
216
|
+
const CURRENCY_LABEL = "currency";
|
|
217
|
+
const QUANTILE_LABEL = "quantile";
|
|
218
|
+
const BUDGET_SECONDS = { hour: 3600, day: 86400 } as const;
|
|
219
|
+
const BUDGET_RANGE = { hour: "1h", day: "1d" } as const;
|
|
220
|
+
|
|
221
|
+
const DEFAULTS = {
|
|
222
|
+
errorRatio: 0.05,
|
|
223
|
+
latencySeconds: 30,
|
|
224
|
+
toolErrorRatio: 0.1,
|
|
225
|
+
for: "10m",
|
|
226
|
+
severity: "warning",
|
|
227
|
+
};
|
|
228
|
+
|
|
229
|
+
function fail(message: string): never {
|
|
230
|
+
throw new Error(`GenAiRules: ${message}`);
|
|
231
|
+
}
|
|
232
|
+
|
|
233
|
+
/** A number as PromQL writes it, without float noise. */
|
|
234
|
+
function num(n: number): string {
|
|
235
|
+
return String(Number(n.toPrecision(12)));
|
|
236
|
+
}
|
|
237
|
+
|
|
238
|
+
/** A label value inside a PromQL string literal. */
|
|
239
|
+
function quote(value: string): string {
|
|
240
|
+
return JSON.stringify(value);
|
|
241
|
+
}
|
|
242
|
+
|
|
243
|
+
type Matcher = [label: string, op: "=" | "!=" | "=~", value: string];
|
|
244
|
+
|
|
245
|
+
function selector(metric: string, matchers: Matcher[]): string {
|
|
246
|
+
if (matchers.length === 0) return metric;
|
|
247
|
+
return `${metric}{${matchers.map(([l, op, v]) => `${l}${op}${quote(v)}`).join(", ")}}`;
|
|
248
|
+
}
|
|
249
|
+
|
|
250
|
+
function sumBy(labels: string[], inner: string): string {
|
|
251
|
+
return `sum by (${labels.join(", ")}) (${inner})`;
|
|
252
|
+
}
|
|
253
|
+
|
|
254
|
+
function metricsOf(genAi: GenAiRulesProps["genAi"]): GenAiMetrics {
|
|
255
|
+
const m = (genAi as { metrics?: GenAiMetrics })?.metrics ?? (genAi as GenAiMetrics);
|
|
256
|
+
if (!m || !m.calls || !m.duration || !m.inputTokens || !m.outputTokens) {
|
|
257
|
+
fail("genAi must be genAiMetrics(...) or genAiComponents(...) from the otel lexicon");
|
|
258
|
+
}
|
|
259
|
+
return m;
|
|
260
|
+
}
|
|
261
|
+
|
|
262
|
+
function has(metric: GenAiMetric, attribute: string): boolean {
|
|
263
|
+
return metric.dimensions.includes(attribute);
|
|
264
|
+
}
|
|
265
|
+
|
|
266
|
+
function need(metric: GenAiMetric, attribute: string): string {
|
|
267
|
+
if (!has(metric, attribute)) fail(`${metric.name} has no ${attribute} attribute, so the rules can't be split by it`);
|
|
268
|
+
return prometheusLabel(attribute);
|
|
269
|
+
}
|
|
270
|
+
|
|
271
|
+
function seconds(metric: GenAiMetric): void {
|
|
272
|
+
if (metric.type !== "histogram") fail(`${metric.name} is a ${metric.type}, not a histogram`);
|
|
273
|
+
if (metric.unit !== "s") fail(`${metric.name} is in ${metric.unit ?? "no unit"}; the latency rules read seconds`);
|
|
274
|
+
}
|
|
275
|
+
|
|
276
|
+
/** How the chosen source spells requests, errors, durations and tokens. */
|
|
277
|
+
interface Source {
|
|
278
|
+
name: "client" | "spans";
|
|
279
|
+
/** The counter whose rate is requests. */
|
|
280
|
+
requests: string;
|
|
281
|
+
/** Matchers that select errors among requests. */
|
|
282
|
+
errorMatchers: Matcher[];
|
|
283
|
+
buckets: string;
|
|
284
|
+
modelLabels: string[];
|
|
285
|
+
provider?: string;
|
|
286
|
+
/** Token rate expression per type, already summed by `tokenLabels`. */
|
|
287
|
+
tokenExpr: (type: string, by: string[], window: string) => string;
|
|
288
|
+
tokenLabels: string[];
|
|
289
|
+
/** True when token series carry the provider label. */
|
|
290
|
+
tokensHaveProvider: boolean;
|
|
291
|
+
}
|
|
292
|
+
|
|
293
|
+
const A = GENAI_ATTRIBUTES;
|
|
294
|
+
|
|
295
|
+
function clientSource(m: GenAiMetrics): Source {
|
|
296
|
+
const c = m.client;
|
|
297
|
+
if (!c) fail('source "client" needs genAiMetrics({ clientMetrics: "derive" | "passthrough" })');
|
|
298
|
+
const d = c.operationDuration;
|
|
299
|
+
const t = c.tokenUsage;
|
|
300
|
+
seconds(d);
|
|
301
|
+
const provider = need(d, A.providerName);
|
|
302
|
+
const model = need(d, A.requestModel);
|
|
303
|
+
const operation = need(d, A.operationName);
|
|
304
|
+
const errorType = need(d, A.errorType);
|
|
305
|
+
const tokenType = need(t, A.tokenType);
|
|
306
|
+
const tokenLabels = [need(t, A.providerName), need(t, A.requestModel)];
|
|
307
|
+
return {
|
|
308
|
+
name: "client",
|
|
309
|
+
requests: `${d.prometheus}_count`,
|
|
310
|
+
errorMatchers: [[errorType, "!=", ""]],
|
|
311
|
+
buckets: `${d.prometheus}_bucket`,
|
|
312
|
+
modelLabels: [provider, model, operation],
|
|
313
|
+
provider,
|
|
314
|
+
tokenExpr: (type, by, window) => sumBy(by, `rate(${selector(`${t.prometheus}_sum`, [[tokenType, "=", type]])}[${window}])`),
|
|
315
|
+
tokenLabels,
|
|
316
|
+
tokensHaveProvider: true,
|
|
317
|
+
};
|
|
318
|
+
}
|
|
319
|
+
|
|
320
|
+
function spanSource(m: GenAiMetrics): Source {
|
|
321
|
+
seconds(m.duration);
|
|
322
|
+
const status = need(m.calls, "status.code");
|
|
323
|
+
const model = need(m.calls, A.requestModel);
|
|
324
|
+
const operation = need(m.calls, A.operationName);
|
|
325
|
+
need(m.calls, A.errorType);
|
|
326
|
+
for (const a of [A.requestModel, A.operationName]) need(m.duration, a);
|
|
327
|
+
const provider = has(m.calls, A.providerName) && has(m.duration, A.providerName) ? prometheusLabel(A.providerName) : undefined;
|
|
328
|
+
const tokenModel = need(m.inputTokens, A.requestModel);
|
|
329
|
+
need(m.outputTokens, A.requestModel);
|
|
330
|
+
const tokensHaveProvider = has(m.inputTokens, A.providerName) && has(m.outputTokens, A.providerName);
|
|
331
|
+
const tokenLabels = tokensHaveProvider ? [prometheusLabel(A.providerName), tokenModel] : [tokenModel];
|
|
332
|
+
const byType: Record<string, GenAiMetric> = { [GENAI_TOKEN_TYPES.input]: m.inputTokens, [GENAI_TOKEN_TYPES.output]: m.outputTokens };
|
|
333
|
+
return {
|
|
334
|
+
name: "spans",
|
|
335
|
+
requests: m.calls.prometheus,
|
|
336
|
+
errorMatchers: [[status, "=", SPAN_STATUS_ERROR]],
|
|
337
|
+
buckets: `${m.duration.prometheus}_bucket`,
|
|
338
|
+
modelLabels: provider ? [provider, model, operation] : [model, operation],
|
|
339
|
+
provider,
|
|
340
|
+
tokenExpr: (type, by, window) => sumBy(by, `rate(${byType[type].prometheus}[${window}])`),
|
|
341
|
+
tokenLabels,
|
|
342
|
+
tokensHaveProvider,
|
|
343
|
+
};
|
|
344
|
+
}
|
|
345
|
+
|
|
346
|
+
interface Resolved {
|
|
347
|
+
props: GenAiRulesProps;
|
|
348
|
+
metrics: GenAiMetrics;
|
|
349
|
+
src: Source;
|
|
350
|
+
prefix: string;
|
|
351
|
+
window: string;
|
|
352
|
+
groupBy: string[];
|
|
353
|
+
/** Tool label when the span metrics carry one. */
|
|
354
|
+
tool?: { label: string; status: string };
|
|
355
|
+
prices: GenAiPrice[];
|
|
356
|
+
}
|
|
357
|
+
|
|
358
|
+
function checkAlertOptions(at: string, a: GenAiAlertOptions): void {
|
|
359
|
+
if (a.for !== undefined && !isValidDuration(a.for)) fail(`${at}.for "${a.for}" is not a Prometheus duration`);
|
|
360
|
+
if (a.severity !== undefined && (typeof a.severity !== "string" || a.severity === "")) fail(`${at}.severity must be a non-empty string`);
|
|
361
|
+
}
|
|
362
|
+
|
|
363
|
+
function checkRatio(at: string, value: unknown): void {
|
|
364
|
+
if (!(typeof value === "number" && value > 0 && value < 1)) fail(`${at} must be above 0 and below 1, got ${String(value)}`);
|
|
365
|
+
}
|
|
366
|
+
|
|
367
|
+
function resolve(props: GenAiRulesProps): Resolved {
|
|
368
|
+
if (!props || typeof props !== "object") fail("props are required");
|
|
369
|
+
const metrics = metricsOf(props.genAi);
|
|
370
|
+
const sourceName = props.source ?? (metrics.client ? "client" : "spans");
|
|
371
|
+
if (sourceName !== "client" && sourceName !== "spans") fail(`source must be "client" or "spans", got ${JSON.stringify(sourceName)}`);
|
|
372
|
+
const src = sourceName === "client" ? clientSource(metrics) : spanSource(metrics);
|
|
373
|
+
|
|
374
|
+
const prefix = props.prefix ?? "gen_ai";
|
|
375
|
+
if (!PREFIX.test(prefix)) fail(`prefix "${prefix}" must be letters, digits and '_', not starting with a digit`);
|
|
376
|
+
const window = props.rateWindow ?? "5m";
|
|
377
|
+
if (!isValidDuration(window) || !durationMs(window)) fail(`rateWindow "${window}" is not a positive Prometheus duration`);
|
|
378
|
+
const name = props.name ?? "genai";
|
|
379
|
+
if (!GROUP_NAME.test(name)) fail(`name "${name}" must be letters, digits, '.', '_' or '-'`);
|
|
380
|
+
const groupBy = props.groupBy ?? [];
|
|
381
|
+
for (const l of groupBy) if (!LABEL.test(l)) fail(`groupBy label "${l}" is not a Prometheus label name`);
|
|
382
|
+
|
|
383
|
+
let tool: Resolved["tool"];
|
|
384
|
+
if (has(metrics.calls, A.toolName) && has(metrics.duration, A.toolName) && has(metrics.calls, "status.code")) {
|
|
385
|
+
seconds(metrics.duration);
|
|
386
|
+
tool = { label: prometheusLabel(A.toolName), status: prometheusLabel("status.code") };
|
|
387
|
+
}
|
|
388
|
+
|
|
389
|
+
const prices = props.prices ?? [];
|
|
390
|
+
if (!Array.isArray(prices)) fail("prices must be a list");
|
|
391
|
+
const seen = new Map<string, string>();
|
|
392
|
+
prices.forEach((p, i) => {
|
|
393
|
+
const at = `prices[${i}]`;
|
|
394
|
+
for (const k of ["provider", "model", "currency", "source"] as const) {
|
|
395
|
+
if (typeof p?.[k] !== "string" || p[k].trim() === "") fail(`${at}.${k} is required`);
|
|
396
|
+
}
|
|
397
|
+
for (const k of ["inputPerMTok", "outputPerMTok"] as const) {
|
|
398
|
+
const v = p[k];
|
|
399
|
+
if (!(typeof v === "number" && Number.isFinite(v) && v >= 0)) fail(`${at}.${k} must be a number at or above 0, got ${String(v)}`);
|
|
400
|
+
}
|
|
401
|
+
if (p.asOf !== undefined && !DATE.test(p.asOf)) fail(`${at}.asOf must be a date, YYYY-MM-DD, got ${JSON.stringify(p.asOf)}`);
|
|
402
|
+
// Without a provider on the token series, one model priced twice can't be told apart.
|
|
403
|
+
const k = src.tokensHaveProvider ? `${p.provider}\u0000${p.model}` : p.model;
|
|
404
|
+
const before = seen.get(k);
|
|
405
|
+
if (before !== undefined) {
|
|
406
|
+
fail(
|
|
407
|
+
src.tokensHaveProvider
|
|
408
|
+
? `${at} prices ${p.provider} ${p.model} again (first at ${before})`
|
|
409
|
+
: `${at} prices model ${p.model} again (first at ${before}); the ${src.name} token metrics have no provider to tell them apart`,
|
|
410
|
+
);
|
|
411
|
+
}
|
|
412
|
+
seen.set(k, at);
|
|
413
|
+
});
|
|
414
|
+
|
|
415
|
+
const alerts = props.alerts ?? {};
|
|
416
|
+
const ratioAlert = (key: "errorRatio" | "toolErrorRatio") => {
|
|
417
|
+
const a = alerts[key];
|
|
418
|
+
if (a === undefined) return;
|
|
419
|
+
if (a !== true) {
|
|
420
|
+
checkAlertOptions(`alerts.${key}`, a);
|
|
421
|
+
if (a.threshold !== undefined) checkRatio(`alerts.${key}.threshold`, a.threshold);
|
|
422
|
+
}
|
|
423
|
+
};
|
|
424
|
+
ratioAlert("errorRatio");
|
|
425
|
+
ratioAlert("toolErrorRatio");
|
|
426
|
+
if (alerts.toolErrorRatio !== undefined && !tool) fail(`alerts.toolErrorRatio needs span metrics with a ${A.toolName} attribute`);
|
|
427
|
+
if (alerts.latency !== undefined && alerts.latency !== true) {
|
|
428
|
+
checkAlertOptions("alerts.latency", alerts.latency);
|
|
429
|
+
const s = alerts.latency.thresholdSeconds;
|
|
430
|
+
if (s !== undefined && !(typeof s === "number" && Number.isFinite(s) && s > 0)) fail(`alerts.latency.thresholdSeconds must be above 0, got ${String(s)}`);
|
|
431
|
+
}
|
|
432
|
+
const currencies = new Set(prices.map((p) => p.currency));
|
|
433
|
+
const budgets = new Map<string, string>();
|
|
434
|
+
if (alerts.budgets !== undefined && !Array.isArray(alerts.budgets)) fail("alerts.budgets must be a list");
|
|
435
|
+
(alerts.budgets ?? []).forEach((b, i) => {
|
|
436
|
+
const at = `alerts.budgets[${i}]`;
|
|
437
|
+
const k = `${b?.per} ${b?.currency}`;
|
|
438
|
+
if (budgets.has(k)) fail(`${at} sets a second budget per ${b.per} in ${b.currency} (first at ${budgets.get(k)})`);
|
|
439
|
+
budgets.set(k, at);
|
|
440
|
+
checkAlertOptions(at, b);
|
|
441
|
+
if (!(typeof b.amount === "number" && Number.isFinite(b.amount) && b.amount > 0)) fail(`${at}.amount must be above 0, got ${String(b.amount)}`);
|
|
442
|
+
if (b.per !== "hour" && b.per !== "day") fail(`${at}.per must be "hour" or "day", got ${JSON.stringify(b.per)}`);
|
|
443
|
+
if (!currencies.has(b.currency)) {
|
|
444
|
+
fail(`${at}.currency ${JSON.stringify(b.currency)} is the currency of no price, so nothing could be spent in it`);
|
|
445
|
+
}
|
|
446
|
+
});
|
|
447
|
+
|
|
448
|
+
return { props, metrics, src, prefix, window, groupBy, tool, prices };
|
|
449
|
+
}
|
|
450
|
+
|
|
451
|
+
function metricsFor(r: Resolved): GenAiRuleMetrics {
|
|
452
|
+
const { prefix: p, window: w, src, groupBy, tool, prices, props } = r;
|
|
453
|
+
const rec = (what: string) => `${p}:${what}:rate${w}`;
|
|
454
|
+
const alerts = props.alerts ?? {};
|
|
455
|
+
const info: GenAiAlertInfo[] = [];
|
|
456
|
+
const sev = (a: true | GenAiAlertOptions) => (a === true ? undefined : a.severity) ?? DEFAULTS.severity;
|
|
457
|
+
if (alerts.errorRatio !== undefined) {
|
|
458
|
+
const a = alerts.errorRatio;
|
|
459
|
+
info.push({ alert: "GenAiErrorRatioHigh", kind: "errorRatio", severity: sev(a), threshold: (a === true ? undefined : a.threshold) ?? DEFAULTS.errorRatio });
|
|
460
|
+
}
|
|
461
|
+
if (alerts.latency !== undefined) {
|
|
462
|
+
const a = alerts.latency;
|
|
463
|
+
info.push({ alert: "GenAiLatencyHigh", kind: "latency", severity: sev(a), threshold: (a === true ? undefined : a.thresholdSeconds) ?? DEFAULTS.latencySeconds });
|
|
464
|
+
}
|
|
465
|
+
if (alerts.toolErrorRatio !== undefined) {
|
|
466
|
+
const a = alerts.toolErrorRatio;
|
|
467
|
+
info.push({ alert: "GenAiToolErrorRatioHigh", kind: "toolErrorRatio", severity: sev(a), threshold: (a === true ? undefined : a.threshold) ?? DEFAULTS.toolErrorRatio });
|
|
468
|
+
}
|
|
469
|
+
for (const b of alerts.budgets ?? []) {
|
|
470
|
+
info.push({ alert: "GenAiSpendOverBudget", kind: "budget", severity: b.severity ?? DEFAULTS.severity, threshold: b.amount, currency: b.currency, per: b.per });
|
|
471
|
+
}
|
|
472
|
+
return {
|
|
473
|
+
group: props.name ?? "genai",
|
|
474
|
+
source: src.name,
|
|
475
|
+
rateWindow: w,
|
|
476
|
+
labels: {
|
|
477
|
+
...(src.provider ? { provider: src.provider } : {}),
|
|
478
|
+
model: prometheusLabel(A.requestModel),
|
|
479
|
+
operation: prometheusLabel(A.operationName),
|
|
480
|
+
errorType: prometheusLabel(A.errorType),
|
|
481
|
+
tokenType: prometheusLabel(A.tokenType),
|
|
482
|
+
...(tool ? { tool: tool.label } : {}),
|
|
483
|
+
currency: CURRENCY_LABEL,
|
|
484
|
+
quantile: QUANTILE_LABEL,
|
|
485
|
+
},
|
|
486
|
+
modelLabels: [...src.modelLabels, ...groupBy],
|
|
487
|
+
tokenLabels: [...src.tokenLabels, ...groupBy],
|
|
488
|
+
requests: rec("requests"),
|
|
489
|
+
errors: rec("errors"),
|
|
490
|
+
errorRatio: rec("error_ratio"),
|
|
491
|
+
errorRatioByType: rec("error_ratio_by_type"),
|
|
492
|
+
latency: { record: `${p}:operation_duration_seconds:quantile_rate${w}`, quantiles: [...QUANTILES] },
|
|
493
|
+
tokens: rec("tokens"),
|
|
494
|
+
...(prices.length > 0 ? { cost: rec("cost") } : {}),
|
|
495
|
+
...(tool
|
|
496
|
+
? {
|
|
497
|
+
tool: {
|
|
498
|
+
calls: rec("tool_calls"),
|
|
499
|
+
errors: rec("tool_errors"),
|
|
500
|
+
errorRatio: rec("tool_error_ratio"),
|
|
501
|
+
latency: { record: `${p}:tool_duration_seconds:quantile_rate${w}`, quantiles: [ALERT_QUANTILE] },
|
|
502
|
+
},
|
|
503
|
+
}
|
|
504
|
+
: {}),
|
|
505
|
+
prices: prices.map((x) => ({ provider: x.provider, model: x.model, currency: x.currency, source: x.source, ...(x.asOf ? { asOf: x.asOf } : {}) })),
|
|
506
|
+
currencies: [...new Set(prices.map((x) => x.currency))],
|
|
507
|
+
alerts: info,
|
|
508
|
+
};
|
|
509
|
+
}
|
|
510
|
+
|
|
511
|
+
function recordingRules(r: Resolved, m: GenAiRuleMetrics): RecordingRule[] {
|
|
512
|
+
const { src, window: w, metrics, tool, prices } = r;
|
|
513
|
+
const model = m.labels.model;
|
|
514
|
+
const errorType = m.labels.errorType;
|
|
515
|
+
const tokenType = m.labels.tokenType;
|
|
516
|
+
// Spans without a model (tool calls, in-process agent steps) aren't model requests.
|
|
517
|
+
const scope: Matcher[] = [[model, "!=", ""]];
|
|
518
|
+
const rate = (metric: string, matchers: Matcher[]) => `rate(${selector(metric, matchers)}[${w}])`;
|
|
519
|
+
const rules: RecordingRule[] = [];
|
|
520
|
+
|
|
521
|
+
rules.push({ record: m.requests, expr: sumBy(m.modelLabels, rate(src.requests, scope)) });
|
|
522
|
+
rules.push({ record: m.errors, expr: sumBy([...m.modelLabels, errorType], rate(src.requests, [...scope, ...src.errorMatchers])) });
|
|
523
|
+
// `or 0 * requests` makes the ratio 0, not absent, for a model with no errors yet.
|
|
524
|
+
rules.push({ record: m.errorRatio, expr: `(\n sum without (${errorType}) (${m.errors})\n or\n 0 * ${m.requests}\n)\n/\n${m.requests}` });
|
|
525
|
+
rules.push({ record: m.errorRatioByType, expr: `${m.errors}\n/ ignoring (${errorType}) group_left\n${m.requests}` });
|
|
526
|
+
for (const q of QUANTILES) {
|
|
527
|
+
rules.push({
|
|
528
|
+
record: m.latency.record,
|
|
529
|
+
expr: `histogram_quantile(${q}, ${sumBy([...m.modelLabels, "le"], rate(src.buckets, scope))})`,
|
|
530
|
+
labels: { [QUANTILE_LABEL]: q },
|
|
531
|
+
});
|
|
532
|
+
}
|
|
533
|
+
for (const type of [GENAI_TOKEN_TYPES.input, GENAI_TOKEN_TYPES.output]) {
|
|
534
|
+
rules.push({ record: m.tokens, expr: src.tokenExpr(type, m.tokenLabels, w), labels: { [tokenType]: type } });
|
|
535
|
+
}
|
|
536
|
+
|
|
537
|
+
if (m.cost) {
|
|
538
|
+
const provider = prometheusLabel(A.providerName);
|
|
539
|
+
for (const price of prices) {
|
|
540
|
+
const of = (type: string, perMTok: number) => {
|
|
541
|
+
const ms: Matcher[] = [[model, "=", price.model], [tokenType, "=", type]];
|
|
542
|
+
if (src.tokensHaveProvider) ms.unshift([provider, "=", price.provider]);
|
|
543
|
+
return `${selector(m.tokens, ms)} * ${num(perMTok)} / 1000000`;
|
|
544
|
+
};
|
|
545
|
+
rules.push({
|
|
546
|
+
record: m.cost,
|
|
547
|
+
expr: `${of(GENAI_TOKEN_TYPES.input, price.inputPerMTok)}\nor\n${of(GENAI_TOKEN_TYPES.output, price.outputPerMTok)}`,
|
|
548
|
+
// Provider and model as rule labels too: they tell the cost rules apart (PROM102), and
|
|
549
|
+
// carry the provider onto span token sums, which have none.
|
|
550
|
+
labels: { [provider]: price.provider, [model]: price.model, [CURRENCY_LABEL]: price.currency },
|
|
551
|
+
});
|
|
552
|
+
}
|
|
553
|
+
}
|
|
554
|
+
|
|
555
|
+
if (tool && m.tool) {
|
|
556
|
+
const t = m.tool;
|
|
557
|
+
const toolScope: Matcher[] = [[tool.label, "!=", ""]];
|
|
558
|
+
const by = [tool.label, ...r.groupBy];
|
|
559
|
+
rules.push({ record: t.calls, expr: sumBy(by, rate(metrics.calls.prometheus, toolScope)) });
|
|
560
|
+
rules.push({ record: t.errors, expr: sumBy(by, rate(metrics.calls.prometheus, [...toolScope, [tool.status, "=", SPAN_STATUS_ERROR]])) });
|
|
561
|
+
rules.push({ record: t.errorRatio, expr: `(\n ${t.errors}\n or\n 0 * ${t.calls}\n)\n/\n${t.calls}` });
|
|
562
|
+
rules.push({
|
|
563
|
+
record: t.latency.record,
|
|
564
|
+
expr: `histogram_quantile(${ALERT_QUANTILE}, ${sumBy([...by, "le"], rate(`${metrics.duration.prometheus}_bucket`, toolScope))})`,
|
|
565
|
+
labels: { [QUANTILE_LABEL]: ALERT_QUANTILE },
|
|
566
|
+
});
|
|
567
|
+
}
|
|
568
|
+
return rules;
|
|
569
|
+
}
|
|
570
|
+
|
|
571
|
+
function alertRules(r: Resolved, m: GenAiRuleMetrics): AlertingRule[] {
|
|
572
|
+
const alerts = r.props.alerts ?? {};
|
|
573
|
+
const l = m.labels;
|
|
574
|
+
const who = `{{ $labels.${l.model} }}${l.provider ? ` ({{ $labels.${l.provider} }})` : ""} {{ $labels.${l.operation} }}`;
|
|
575
|
+
const out: AlertingRule[] = [];
|
|
576
|
+
const build = (a: true | GenAiAlertOptions, info: GenAiAlertInfo, expr: string, summary: string, description: string, extra: LabelSet = {}) => {
|
|
577
|
+
const o: GenAiAlertOptions = a === true ? {} : a;
|
|
578
|
+
out.push({
|
|
579
|
+
alert: info.alert,
|
|
580
|
+
expr,
|
|
581
|
+
for: o.for ?? DEFAULTS.for,
|
|
582
|
+
labels: { ...(o.labels ?? {}), ...extra, severity: info.severity },
|
|
583
|
+
annotations: { summary, description, ...(o.annotations ?? {}) },
|
|
584
|
+
});
|
|
585
|
+
};
|
|
586
|
+
const byKind = (k: GenAiAlertInfo["kind"]) => m.alerts.filter((x) => x.kind === k);
|
|
587
|
+
|
|
588
|
+
if (alerts.errorRatio !== undefined) {
|
|
589
|
+
const info = byKind("errorRatio")[0];
|
|
590
|
+
build(
|
|
591
|
+
alerts.errorRatio,
|
|
592
|
+
info,
|
|
593
|
+
`${m.errorRatio} > ${num(info.threshold)}`,
|
|
594
|
+
`GenAI error ratio above ${num(info.threshold * 100)}% for ${who}`,
|
|
595
|
+
`{{ $value | humanizePercentage }} of ${who} requests failed over the last ${r.window}.`,
|
|
596
|
+
);
|
|
597
|
+
}
|
|
598
|
+
if (alerts.latency !== undefined) {
|
|
599
|
+
const info = byKind("latency")[0];
|
|
600
|
+
build(
|
|
601
|
+
alerts.latency,
|
|
602
|
+
info,
|
|
603
|
+
`${selector(m.latency.record, [[QUANTILE_LABEL, "=", ALERT_QUANTILE]])} > ${num(info.threshold)}`,
|
|
604
|
+
`GenAI p95 latency above ${num(info.threshold)}s for ${who}`,
|
|
605
|
+
`p95 latency of ${who} is {{ $value | humanizeDuration }} over the last ${r.window}.`,
|
|
606
|
+
);
|
|
607
|
+
}
|
|
608
|
+
if (alerts.toolErrorRatio !== undefined && m.tool) {
|
|
609
|
+
const info = byKind("toolErrorRatio")[0];
|
|
610
|
+
const tool = `{{ $labels.${l.tool} }}`;
|
|
611
|
+
build(
|
|
612
|
+
alerts.toolErrorRatio,
|
|
613
|
+
info,
|
|
614
|
+
`${m.tool.errorRatio} > ${num(info.threshold)}`,
|
|
615
|
+
`Tool ${tool} error ratio above ${num(info.threshold * 100)}%`,
|
|
616
|
+
`{{ $value | humanizePercentage }} of calls to tool ${tool} failed over the last ${r.window}.`,
|
|
617
|
+
);
|
|
618
|
+
}
|
|
619
|
+
(alerts.budgets ?? []).forEach((b, i) => {
|
|
620
|
+
const info = byKind("budget")[i];
|
|
621
|
+
// The average of the recorded per-second rate over the window, times its length, is what the window cost.
|
|
622
|
+
const spend = `sum by (${CURRENCY_LABEL}) (avg_over_time(${selector(m.cost!, [[CURRENCY_LABEL, "=", b.currency]])}[${BUDGET_RANGE[b.per]}])) * ${BUDGET_SECONDS[b.per]}`;
|
|
623
|
+
build(
|
|
624
|
+
{ for: "0s", ...b },
|
|
625
|
+
info,
|
|
626
|
+
`${spend} > ${num(b.amount)}`,
|
|
627
|
+
`GenAI spend over the ${b.per === "hour" ? "hourly" : "daily"} budget of ${num(b.amount)} ${b.currency}`,
|
|
628
|
+
`Model calls cost {{ $value | printf "%.2f" }} ${b.currency} over the last ${BUDGET_RANGE[b.per]}, above the budget of ${num(b.amount)}. Models without a price are not counted.`,
|
|
629
|
+
{ budget: b.per, [CURRENCY_LABEL]: b.currency },
|
|
630
|
+
);
|
|
631
|
+
});
|
|
632
|
+
// A for of 0s says nothing; leave it out.
|
|
633
|
+
for (const rule of out) if (rule.for === "0s") delete rule.for;
|
|
634
|
+
return out;
|
|
635
|
+
}
|
|
636
|
+
|
|
637
|
+
/**
|
|
638
|
+
* Recording rules and opt-in alerts for GenAI calls, from the otel preset's metrics.
|
|
639
|
+
*
|
|
640
|
+
* @example
|
|
641
|
+
* ```ts
|
|
642
|
+
* import { genAiMetrics } from "@intentius/chant-lexicon-otel";
|
|
643
|
+
* import { GenAiRules } from "@intentius/chant-lexicon-prometheus";
|
|
644
|
+
*
|
|
645
|
+
* export const genai = GenAiRules({
|
|
646
|
+
* genAi: genAiMetrics({ clientMetrics: "derive" }),
|
|
647
|
+
* prices: [{ provider: "anthropic", model: "claude-x", inputPerMTok: 3, outputPerMTok: 15, currency: "USD", source: "https://example.com/pricing", asOf: "2026-09-29" }],
|
|
648
|
+
* alerts: { errorRatio: true, budgets: [{ amount: 50, currency: "USD", per: "day" }] },
|
|
649
|
+
* });
|
|
650
|
+
* // genai.rules is a RuleGroup; genAiRuleMetrics(genai) names its series.
|
|
651
|
+
* ```
|
|
652
|
+
*/
|
|
653
|
+
export const GenAiRules = Composite<GenAiRulesProps, GenAiRulesMembers>((props) => {
|
|
654
|
+
const r = resolve(props);
|
|
655
|
+
const m = metricsFor(r);
|
|
656
|
+
const group = new RuleGroup({
|
|
657
|
+
name: m.group,
|
|
658
|
+
...(props.interval !== undefined ? { interval: props.interval } : {}),
|
|
659
|
+
...(props.labels !== undefined ? { labels: props.labels } : {}),
|
|
660
|
+
rules: [...recordingRules(r, m), ...alertRules(r, m)],
|
|
661
|
+
});
|
|
662
|
+
Object.defineProperty(group, GENAI_RULES_METRICS, { value: m, enumerable: false });
|
|
663
|
+
return { rules: group };
|
|
664
|
+
}, "GenAiRules");
|
|
665
|
+
|
|
666
|
+
/**
|
|
667
|
+
* The series a `GenAiRules` records and the alerts it builds. Pass the
|
|
668
|
+
* `GenAiRules(...)` result, its rule group, or the props it was built from;
|
|
669
|
+
* a dashboard reads names from here instead of repeating them.
|
|
670
|
+
*/
|
|
671
|
+
export function genAiRuleMetrics(rules: GenAiRulesInstance | RuleGroupEntity | GenAiRulesProps): GenAiRuleMetrics {
|
|
672
|
+
const stashed = (x: unknown): GenAiRuleMetrics | undefined =>
|
|
673
|
+
typeof x === "object" && x !== null ? ((x as Record<symbol, unknown>)[GENAI_RULES_METRICS] as GenAiRuleMetrics | undefined) : undefined;
|
|
674
|
+
const direct = stashed(rules) ?? stashed((rules as Partial<GenAiRulesMembers>)?.rules);
|
|
675
|
+
if (direct) return structuredClone(direct);
|
|
676
|
+
if (typeof rules === "object" && rules !== null && "genAi" in rules) return metricsFor(resolve(rules as GenAiRulesProps));
|
|
677
|
+
throw new Error("genAiRuleMetrics: pass a GenAiRules(...) result, its rule group, or GenAiRules props");
|
|
678
|
+
}
|