@intentius/chant-lexicon-prometheus 0.96.0 → 0.98.0
This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
- package/README.md +5 -0
- package/dist/alertmanager.d.ts +4 -2
- package/dist/alertmanager.d.ts.map +1 -1
- package/dist/build.d.ts.map +1 -1
- package/dist/codegen/docs.d.ts.map +1 -1
- package/dist/import/generator.d.ts +35 -0
- package/dist/import/generator.d.ts.map +1 -0
- package/dist/import/parser.d.ts +52 -0
- package/dist/import/parser.d.ts.map +1 -0
- package/dist/import/slo.d.ts +15 -0
- package/dist/import/slo.d.ts.map +1 -0
- package/dist/integrity.json +7 -7
- package/dist/lint/rules/literal-credential.d.ts.map +1 -1
- package/dist/lint/rules/prom-ast.d.ts +8 -0
- package/dist/lint/rules/prom-ast.d.ts.map +1 -1
- package/dist/lint/rules/promql-literal.d.ts.map +1 -1
- package/dist/manifest.json +1 -1
- package/dist/model.d.ts +21 -0
- package/dist/model.d.ts.map +1 -1
- package/dist/plugin.d.ts.map +1 -1
- package/dist/rules/literal-credential.ts +17 -11
- package/dist/rules/prom-ast.ts +36 -0
- package/dist/rules/promql-literal.ts +15 -1
- package/dist/skills/chant-prometheus-alertmanager.md +4 -0
- package/dist/skills/chant-prometheus.md +11 -0
- package/package.json +2 -2
- package/src/alertmanager.ts +4 -1
- package/src/build.ts +2 -0
- package/src/codegen/docs.ts +5 -1
- package/src/import/cli.test.ts +98 -0
- package/src/import/generated-types.e2e.test.ts +75 -0
- package/src/import/generator.test.ts +121 -0
- package/src/import/generator.ts +570 -0
- package/src/import/parser.test.ts +120 -0
- package/src/import/parser.ts +348 -0
- package/src/import/roundtrip.test.ts +321 -0
- package/src/import/slo.ts +175 -0
- package/src/import/testdata/alertmanager-full.yml +89 -0
- package/src/import/testdata/fixtures.ts +114 -0
- package/src/import/testdata/rules-full.yml +42 -0
- package/src/import/testdata/upstream/alertmanager-route-labels.yml +65 -0
- package/src/import/testdata/upstream/alertmanager-simple.yml +130 -0
- package/src/import/testdata/upstream/prometheus-alerting-rules.yml +15 -0
- package/src/import/testdata/upstream/prometheus-alerting-templates.yml +23 -0
- package/src/import/testdata/upstream/prometheus-recording-rules.yml +7 -0
- package/src/lint/rules/literal-credential.ts +17 -11
- package/src/lint/rules/prom-ast.ts +36 -0
- package/src/lint/rules/promql-literal.ts +15 -1
- package/src/lint/rules/rules.test.ts +28 -0
- package/src/model.ts +22 -0
- package/src/plugin.ts +10 -0
- package/src/skills/chant-prometheus-alertmanager.md +4 -0
- package/src/skills/chant-prometheus.md +11 -0
|
@@ -0,0 +1,175 @@
|
|
|
1
|
+
/**
|
|
2
|
+
* Recognise a rule group that `Slo()` built, so the importer can write the
|
|
3
|
+
* `Slo` declaration instead of its forty-odd lines of rules.
|
|
4
|
+
*
|
|
5
|
+
* The group's name, recorded series, objective and burn-rate alerts give
|
|
6
|
+
* back candidate props; the candidate is accepted only when `Slo()` builds
|
|
7
|
+
* it to the same group, rule for rule. Anything else (a hand-edited
|
|
8
|
+
* threshold, an extra rule, a different description) fails that check and
|
|
9
|
+
* the group is imported as a plain `RuleGroup`.
|
|
10
|
+
*/
|
|
11
|
+
|
|
12
|
+
import { Slo, DEFAULT_BURN_RATES, SLO_WINDOW_PLACEHOLDER, type SloAlertTier, type SloProps, type BurnRateWindow } from "../composites/slo";
|
|
13
|
+
import { ruleGroupConfig } from "../rules";
|
|
14
|
+
import { durationMs } from "../duration";
|
|
15
|
+
import { isAlertingRuleConfig, isRecordingRuleConfig, type AlertingRuleConfig, type LabelSet, type RuleGroupConfig } from "../model";
|
|
16
|
+
|
|
17
|
+
const RATIO_PREFIX = "slo:sli_error:ratio_rate";
|
|
18
|
+
const OBJECTIVE = "slo:objective:ratio";
|
|
19
|
+
const BUDGET = "slo:error_budget:remaining";
|
|
20
|
+
const DEFAULT_ALERT_NAME = "ErrorBudgetBurn";
|
|
21
|
+
const PAIR_LABELS = new Set(["slo", "severity", "long_window", "short_window"]);
|
|
22
|
+
const DESCRIPTION_TAIL = "The error ratio over both the last ";
|
|
23
|
+
|
|
24
|
+
/** Objects with their keys sorted, so two configs compare equal whatever order their keys were written in. */
|
|
25
|
+
function canonical(v: unknown): unknown {
|
|
26
|
+
if (Array.isArray(v)) return v.map(canonical);
|
|
27
|
+
if (typeof v === "object" && v !== null) {
|
|
28
|
+
return Object.fromEntries(
|
|
29
|
+
Object.keys(v)
|
|
30
|
+
.sort()
|
|
31
|
+
.map((k) => [k, canonical((v as Record<string, unknown>)[k])]),
|
|
32
|
+
);
|
|
33
|
+
}
|
|
34
|
+
return v;
|
|
35
|
+
}
|
|
36
|
+
|
|
37
|
+
function same(a: unknown, b: unknown): boolean {
|
|
38
|
+
return JSON.stringify(canonical(a)) === JSON.stringify(canonical(b));
|
|
39
|
+
}
|
|
40
|
+
|
|
41
|
+
/** The SLI expressions a recorded error ratio was built from, with the window put back as `{{window}}`. */
|
|
42
|
+
function sliFrom(expr: string, window: string): SloProps["sli"] | undefined {
|
|
43
|
+
const unwindow = (s: string) => s.split(`[${window}]`).join(`[${SLO_WINDOW_PLACEHOLDER}]`);
|
|
44
|
+
const goodHead = "1 - (\n (";
|
|
45
|
+
const goodTail = ")\n)";
|
|
46
|
+
if (expr.startsWith(goodHead) && expr.endsWith(goodTail)) {
|
|
47
|
+
const inner = expr.slice(goodHead.length, -goodTail.length);
|
|
48
|
+
const at = inner.indexOf(")\n /\n (");
|
|
49
|
+
if (at === -1) return undefined;
|
|
50
|
+
return { good: unwindow(inner.slice(0, at)), total: unwindow(inner.slice(at + ")\n /\n (".length)) };
|
|
51
|
+
}
|
|
52
|
+
if (expr.startsWith("(") && expr.endsWith(")")) {
|
|
53
|
+
const inner = expr.slice(1, -1);
|
|
54
|
+
const at = inner.indexOf(")\n/\n(");
|
|
55
|
+
if (at === -1) return undefined;
|
|
56
|
+
return { errors: unwindow(inner.slice(0, at)), total: unwindow(inner.slice(at + ")\n/\n(".length)) };
|
|
57
|
+
}
|
|
58
|
+
return undefined;
|
|
59
|
+
}
|
|
60
|
+
|
|
61
|
+
function extras(set: LabelSet | undefined, skip: (k: string) => boolean): LabelSet | undefined {
|
|
62
|
+
const out: LabelSet = {};
|
|
63
|
+
for (const [k, v] of Object.entries(set ?? {})) if (!skip(k)) out[k] = v;
|
|
64
|
+
return Object.keys(out).length > 0 ? out : undefined;
|
|
65
|
+
}
|
|
66
|
+
|
|
67
|
+
function isDefaultPairs(tier: "page" | "ticket", pairs: BurnRateWindow[], factors: number[], windowMs: (w: string) => number, sloWindow: string): boolean {
|
|
68
|
+
const defaults = DEFAULT_BURN_RATES[tier];
|
|
69
|
+
if (defaults.length !== pairs.length) return false;
|
|
70
|
+
return defaults.every((d, i) => {
|
|
71
|
+
const p = pairs[i];
|
|
72
|
+
const factor = Number(((d.budgetConsumed! * windowMs(sloWindow)) / windowMs(d.long)).toPrecision(6));
|
|
73
|
+
return p.long === d.long && p.short === d.short && factor === factors[i];
|
|
74
|
+
});
|
|
75
|
+
}
|
|
76
|
+
|
|
77
|
+
/** The `Slo` props that build exactly `group`, or `undefined` when no `Slo()` call does. */
|
|
78
|
+
export function recognizeSlo(group: RuleGroupConfig): SloProps | undefined {
|
|
79
|
+
if (!group.name.startsWith("slo-") || group.query_offset !== undefined || group.limit !== undefined) return undefined;
|
|
80
|
+
const name = group.name.slice("slo-".length);
|
|
81
|
+
const records = group.rules.filter(isRecordingRuleConfig);
|
|
82
|
+
const alerts = group.rules.filter(isAlertingRuleConfig) as AlertingRuleConfig[];
|
|
83
|
+
if (records.length + alerts.length !== group.rules.length) return undefined;
|
|
84
|
+
|
|
85
|
+
const objectiveAt = records.findIndex((r) => r.record === OBJECTIVE);
|
|
86
|
+
if (objectiveAt < 1 || records[objectiveAt + 1]?.record !== BUDGET) return undefined;
|
|
87
|
+
const objective = Number(/^vector\((.+)\)$/.exec(records[objectiveAt].expr)?.[1]);
|
|
88
|
+
if (!Number.isFinite(objective)) return undefined;
|
|
89
|
+
const windowRecord = records[objectiveAt - 1].record;
|
|
90
|
+
if (!windowRecord.startsWith(RATIO_PREFIX)) return undefined;
|
|
91
|
+
const window = windowRecord.slice(RATIO_PREFIX.length);
|
|
92
|
+
|
|
93
|
+
// Candidate SLIs: from each alert window's ratio, or from the whole-window
|
|
94
|
+
// ratio when there are no alerts.
|
|
95
|
+
const ratioRecords = records.slice(0, objectiveAt - 1);
|
|
96
|
+
const sources = ratioRecords.length > 0 ? ratioRecords : [records[objectiveAt - 1]];
|
|
97
|
+
const slis = sources
|
|
98
|
+
.map((r) => (r.record.startsWith(RATIO_PREFIX) ? sliFrom(r.expr, r.record.slice(RATIO_PREFIX.length)) : undefined))
|
|
99
|
+
.filter((s): s is SloProps["sli"] => s !== undefined);
|
|
100
|
+
if (slis.length === 0) return undefined;
|
|
101
|
+
|
|
102
|
+
// Alert tiers: `page` and `ticket` by severity, or custom severities in the order they appear.
|
|
103
|
+
const alertName = alerts[0]?.alert ?? DEFAULT_ALERT_NAME;
|
|
104
|
+
const bySeverity = new Map<string, AlertingRuleConfig[]>();
|
|
105
|
+
for (const a of alerts) {
|
|
106
|
+
const severity = a.labels?.severity;
|
|
107
|
+
if (severity === undefined || a.alert !== alertName) return undefined;
|
|
108
|
+
bySeverity.set(severity, [...(bySeverity.get(severity) ?? []), a]);
|
|
109
|
+
}
|
|
110
|
+
if (bySeverity.size > 2) return undefined;
|
|
111
|
+
const severities = [...bySeverity.keys()];
|
|
112
|
+
const tierOf = new Map<string, "page" | "ticket">();
|
|
113
|
+
for (const s of severities) if (s === "page" || s === "ticket") tierOf.set(s, s);
|
|
114
|
+
for (const s of severities) {
|
|
115
|
+
if (tierOf.has(s)) continue;
|
|
116
|
+
const free = (["page", "ticket"] as const).find((t) => ![...tierOf.values()].includes(t));
|
|
117
|
+
if (!free) return undefined;
|
|
118
|
+
tierOf.set(s, free);
|
|
119
|
+
}
|
|
120
|
+
|
|
121
|
+
let description: string | undefined;
|
|
122
|
+
const alerting: NonNullable<SloProps["alerting"]> = {};
|
|
123
|
+
for (const tier of ["page", "ticket"] as const) {
|
|
124
|
+
const severity = [...tierOf].find(([, t]) => t === tier)?.[0];
|
|
125
|
+
if (severity === undefined) {
|
|
126
|
+
alerting[tier] = false;
|
|
127
|
+
continue;
|
|
128
|
+
}
|
|
129
|
+
const tierAlerts = bySeverity.get(severity)!;
|
|
130
|
+
const first = tierAlerts[0];
|
|
131
|
+
const pairs: BurnRateWindow[] = [];
|
|
132
|
+
const factors: number[] = [];
|
|
133
|
+
for (const a of tierAlerts) {
|
|
134
|
+
const long = a.labels?.long_window;
|
|
135
|
+
const short = a.labels?.short_window;
|
|
136
|
+
const factor = Number(/> \(([^ ]+) \* /.exec(a.expr)?.[1]);
|
|
137
|
+
if (long === undefined || short === undefined || !Number.isFinite(factor)) return undefined;
|
|
138
|
+
pairs.push({ long, short, factor });
|
|
139
|
+
factors.push(factor);
|
|
140
|
+
}
|
|
141
|
+
const t: SloAlertTier = {};
|
|
142
|
+
t.burnRates = isDefaultPairs(tier, pairs, factors, (w) => durationMs(w) ?? 0, window) ? "default" : pairs;
|
|
143
|
+
if (severity !== tier) t.severity = severity;
|
|
144
|
+
if (first.for !== undefined) t.for = first.for;
|
|
145
|
+
const labels = extras(first.labels, (k) => PAIR_LABELS.has(k));
|
|
146
|
+
if (labels) t.labels = labels;
|
|
147
|
+
const annotations = extras(first.annotations, (k) => k === "summary" || k === "description");
|
|
148
|
+
if (annotations) t.annotations = annotations;
|
|
149
|
+
const desc = first.annotations?.description ?? "";
|
|
150
|
+
const at = desc.indexOf(DESCRIPTION_TAIL);
|
|
151
|
+
if (at > 0 && description === undefined) description = desc.slice(0, at).trimEnd();
|
|
152
|
+
const onlyDefault = t.burnRates === "default" && Object.keys(t).length === 1;
|
|
153
|
+
if (!onlyDefault) alerting[tier] = t;
|
|
154
|
+
}
|
|
155
|
+
if (alertName !== DEFAULT_ALERT_NAME) alerting.alertName = alertName;
|
|
156
|
+
|
|
157
|
+
for (const sli of slis) {
|
|
158
|
+
const props: SloProps = {
|
|
159
|
+
name,
|
|
160
|
+
objective,
|
|
161
|
+
window,
|
|
162
|
+
...(description ? { description } : {}),
|
|
163
|
+
sli,
|
|
164
|
+
...(Object.keys(alerting).length > 0 ? { alerting } : {}),
|
|
165
|
+
...(group.labels ? { labels: group.labels } : {}),
|
|
166
|
+
...(group.interval !== undefined ? { interval: group.interval } : {}),
|
|
167
|
+
};
|
|
168
|
+
try {
|
|
169
|
+
if (same(ruleGroupConfig(Slo(props).rules), ruleGroupConfig(group))) return props;
|
|
170
|
+
} catch {
|
|
171
|
+
// Not buildable from these props; try the next candidate.
|
|
172
|
+
}
|
|
173
|
+
}
|
|
174
|
+
return undefined;
|
|
175
|
+
}
|
|
@@ -0,0 +1,89 @@
|
|
|
1
|
+
# Deprecated spellings (match, match_re, source_match, the top-level
|
|
2
|
+
# mute_time_intervals), integrations this lexicon does not type (opsgenie,
|
|
3
|
+
# msteams), global fields outside the typed set, *_file credentials, Go
|
|
4
|
+
# templates, a YAML anchor with a merge key, and one literal credential.
|
|
5
|
+
global:
|
|
6
|
+
resolve_timeout: 5m
|
|
7
|
+
smtp_smarthost: smtp.example.com:587
|
|
8
|
+
smtp_from: alertmanager@example.com
|
|
9
|
+
smtp_auth_username: alertmanager
|
|
10
|
+
smtp_auth_password_file: /etc/alertmanager/secrets/smtp-password
|
|
11
|
+
slack_api_url_file: /etc/alertmanager/secrets/slack-url
|
|
12
|
+
opsgenie_api_key_file: /etc/alertmanager/secrets/opsgenie-key
|
|
13
|
+
http_config:
|
|
14
|
+
follow_redirects: true
|
|
15
|
+
templates:
|
|
16
|
+
- /etc/alertmanager/templates/*.tmpl
|
|
17
|
+
route:
|
|
18
|
+
receiver: default
|
|
19
|
+
group_by: [alertname, cluster]
|
|
20
|
+
group_wait: 30s
|
|
21
|
+
labels:
|
|
22
|
+
summary: '{{ .GroupLabels.alertname }} in {{ .GroupLabels.cluster }}'
|
|
23
|
+
routes:
|
|
24
|
+
- match:
|
|
25
|
+
severity: page
|
|
26
|
+
match_re:
|
|
27
|
+
service: ^(api|web)$
|
|
28
|
+
receiver: oncall
|
|
29
|
+
continue: true
|
|
30
|
+
- matchers: ['team="db"']
|
|
31
|
+
receiver: db-chat
|
|
32
|
+
mute_time_intervals: [weekends]
|
|
33
|
+
active_time_intervals: [business-hours]
|
|
34
|
+
- match:
|
|
35
|
+
alertname: Watchdog
|
|
36
|
+
receiver: heartbeat
|
|
37
|
+
repeat_interval: 1m
|
|
38
|
+
inhibit_rules:
|
|
39
|
+
- name: page-mutes-ticket
|
|
40
|
+
source_match:
|
|
41
|
+
severity: page
|
|
42
|
+
target_matchers: ['severity="ticket"']
|
|
43
|
+
equal: [alertname, cluster]
|
|
44
|
+
receivers:
|
|
45
|
+
- name: default
|
|
46
|
+
webhook_configs:
|
|
47
|
+
- url_file: /etc/alertmanager/secrets/default-webhook-url
|
|
48
|
+
max_alerts: 10
|
|
49
|
+
- name: oncall
|
|
50
|
+
pagerduty_configs:
|
|
51
|
+
- routing_key_file: /etc/alertmanager/secrets/pagerduty-key
|
|
52
|
+
description: '{{ template "pagerduty.default.description" . }}'
|
|
53
|
+
details:
|
|
54
|
+
firing: '{{ .Alerts.Firing | len }}'
|
|
55
|
+
opsgenie_configs:
|
|
56
|
+
- api_key_file: /etc/alertmanager/secrets/opsgenie-key
|
|
57
|
+
priority: P1
|
|
58
|
+
- name: db-chat
|
|
59
|
+
slack_configs:
|
|
60
|
+
- &slack
|
|
61
|
+
channel: '#db-alerts'
|
|
62
|
+
title: '{{ .CommonLabels.alertname }}'
|
|
63
|
+
text: >-
|
|
64
|
+
{{ range .Alerts }}{{ .Annotations.summary }}
|
|
65
|
+
{{ end }}
|
|
66
|
+
send_resolved: true
|
|
67
|
+
- <<: *slack
|
|
68
|
+
channel: '#db-oncall'
|
|
69
|
+
msteams_configs:
|
|
70
|
+
- webhook_url_file: /etc/alertmanager/secrets/teams-url
|
|
71
|
+
- name: heartbeat
|
|
72
|
+
email_configs:
|
|
73
|
+
- to: heartbeat@example.com
|
|
74
|
+
auth_username: heartbeat
|
|
75
|
+
auth_password: hunter2
|
|
76
|
+
headers:
|
|
77
|
+
Subject: '[heartbeat] {{ .CommonLabels.alertname }}'
|
|
78
|
+
time_intervals:
|
|
79
|
+
- name: business-hours
|
|
80
|
+
time_intervals:
|
|
81
|
+
- weekdays: ['monday:friday']
|
|
82
|
+
times:
|
|
83
|
+
- start_time: '09:00'
|
|
84
|
+
end_time: '17:00'
|
|
85
|
+
location: Europe/Berlin
|
|
86
|
+
mute_time_intervals:
|
|
87
|
+
- name: weekends
|
|
88
|
+
time_intervals:
|
|
89
|
+
- weekdays: [saturday, sunday]
|
|
@@ -0,0 +1,114 @@
|
|
|
1
|
+
/**
|
|
2
|
+
* Fixtures shared by the import round-trip tests (roundtrip.test.ts,
|
|
3
|
+
* cli.test.ts) and the type-check of the generated source
|
|
4
|
+
* (generated-types.e2e.test.ts).
|
|
5
|
+
*/
|
|
6
|
+
import { readdirSync, readFileSync, statSync } from "fs";
|
|
7
|
+
import { join, resolve } from "path";
|
|
8
|
+
import { build } from "@intentius/chant/build";
|
|
9
|
+
import type { Serializer, SerializerResult } from "@intentius/chant/serializer";
|
|
10
|
+
import { k8sSerializer } from "@intentius/chant-lexicon-k8s";
|
|
11
|
+
import { prometheusSerializer, ALERTMANAGER_FILE } from "../../serializer";
|
|
12
|
+
import { ruleFileYaml } from "../../build";
|
|
13
|
+
import { Slo } from "../../composites/slo";
|
|
14
|
+
|
|
15
|
+
export const pkgDir = resolve(import.meta.dirname, "../../..");
|
|
16
|
+
export const repoRoot = resolve(pkgDir, "../..");
|
|
17
|
+
export const read = (...p: string[]) => readFileSync(join(import.meta.dirname, ...p), "utf-8");
|
|
18
|
+
|
|
19
|
+
/** One file a build wrote: the rule file or `alertmanager.yml`. */
|
|
20
|
+
export interface BuiltFile {
|
|
21
|
+
name: string;
|
|
22
|
+
yaml: string;
|
|
23
|
+
}
|
|
24
|
+
|
|
25
|
+
/** The rule file and `alertmanager.yml` of one build output, whichever it has. */
|
|
26
|
+
export function builtFiles(out: string | SerializerResult | undefined): { rules?: string; alertmanager?: string } {
|
|
27
|
+
if (out === undefined || out === "") return {};
|
|
28
|
+
if (typeof out === "string") return out.startsWith("groups:") ? { rules: out } : { alertmanager: out };
|
|
29
|
+
const primaryIsRules = out.primary.startsWith("groups:");
|
|
30
|
+
return {
|
|
31
|
+
...(primaryIsRules ? { rules: out.primary } : { alertmanager: out.primary }),
|
|
32
|
+
...(out.files?.[ALERTMANAGER_FILE] !== undefined ? { alertmanager: out.files[ALERTMANAGER_FILE] } : {}),
|
|
33
|
+
};
|
|
34
|
+
}
|
|
35
|
+
|
|
36
|
+
/** Every file each prometheus example builds, named `<example>/rules.yml` or `<example>/alertmanager.yml`. */
|
|
37
|
+
export async function exampleOutputs(): Promise<BuiltFile[]> {
|
|
38
|
+
const examplesDir = join(pkgDir, "examples");
|
|
39
|
+
const out: BuiltFile[] = [];
|
|
40
|
+
for (const name of readdirSync(examplesDir).sort()) {
|
|
41
|
+
const srcDir = join(examplesDir, name, "src");
|
|
42
|
+
try {
|
|
43
|
+
if (!statSync(srcDir).isDirectory()) continue;
|
|
44
|
+
} catch {
|
|
45
|
+
continue;
|
|
46
|
+
}
|
|
47
|
+
// k3d-stack declares its Kubernetes workloads beside the rules.
|
|
48
|
+
const serializers: Serializer[] = name === "k3d-stack" ? [k8sSerializer, prometheusSerializer] : [prometheusSerializer];
|
|
49
|
+
const result = await build(srcDir, serializers);
|
|
50
|
+
if (result.errors.length > 0) throw new Error(`${name}: ${result.errors.map(String).join("; ")}`);
|
|
51
|
+
const files = builtFiles(result.outputs.get("prometheus"));
|
|
52
|
+
if (files.rules !== undefined) out.push({ name: `${name}/rules.yml`, yaml: files.rules });
|
|
53
|
+
if (files.alertmanager !== undefined) out.push({ name: `${name}/alertmanager.yml`, yaml: files.alertmanager });
|
|
54
|
+
}
|
|
55
|
+
return out;
|
|
56
|
+
}
|
|
57
|
+
|
|
58
|
+
const calls = "traces_span_metrics_calls_total";
|
|
59
|
+
|
|
60
|
+
/** Rule files `Slo()` builds, over the props that change its shape. */
|
|
61
|
+
export function sloOutputs(): BuiltFile[] {
|
|
62
|
+
const good = {
|
|
63
|
+
good: `sum(rate(${calls}{span_name="checkout",status_code!="STATUS_CODE_ERROR"}[{{window}}]))`,
|
|
64
|
+
total: `sum(rate(${calls}{span_name="checkout"}[{{window}}]))`,
|
|
65
|
+
};
|
|
66
|
+
const errors = {
|
|
67
|
+
errors: `sum(rate(http_requests_total{code=~"5.."}[{{window}}]))`,
|
|
68
|
+
total: "sum(rate(http_requests_total[{{window}}]))",
|
|
69
|
+
};
|
|
70
|
+
const slos = [
|
|
71
|
+
Slo({ name: "defaults", objective: 0.999, window: "30d", sli: errors }),
|
|
72
|
+
Slo({
|
|
73
|
+
name: "order-ack",
|
|
74
|
+
objective: 0.995,
|
|
75
|
+
window: "28d",
|
|
76
|
+
description: "Orders are acknowledged without an error span.",
|
|
77
|
+
sli: good,
|
|
78
|
+
alerting: {
|
|
79
|
+
page: { burnRates: "default", annotations: { runbook_url: "https://runbooks.example.com/order-ack" } },
|
|
80
|
+
ticket: { burnRates: "default", for: "15m", labels: { team: "orders" } },
|
|
81
|
+
},
|
|
82
|
+
labels: { team: "orders" },
|
|
83
|
+
interval: "1m",
|
|
84
|
+
}),
|
|
85
|
+
Slo({
|
|
86
|
+
name: "custom",
|
|
87
|
+
objective: 0.99,
|
|
88
|
+
window: "7d",
|
|
89
|
+
sli: errors,
|
|
90
|
+
alerting: {
|
|
91
|
+
alertName: "LatencyBudgetBurn",
|
|
92
|
+
page: {
|
|
93
|
+
severity: "critical",
|
|
94
|
+
burnRates: [
|
|
95
|
+
{ long: "1h", short: "5m", factor: 10 },
|
|
96
|
+
{ long: "6h", short: "30m", budgetConsumed: 0.1 },
|
|
97
|
+
],
|
|
98
|
+
},
|
|
99
|
+
ticket: false,
|
|
100
|
+
},
|
|
101
|
+
}),
|
|
102
|
+
Slo({ name: "no-alerts", objective: 0.95, window: "1w", sli: good, alerting: { page: false, ticket: false } }),
|
|
103
|
+
];
|
|
104
|
+
return slos.map((s) => ({ name: `Slo ${s.rules.props.name}`, yaml: ruleFileYaml([s.rules]) }));
|
|
105
|
+
}
|
|
106
|
+
|
|
107
|
+
/** The vendored upstream samples, each with the source it came from in its header. */
|
|
108
|
+
export const UPSTREAM = [
|
|
109
|
+
"prometheus-recording-rules.yml",
|
|
110
|
+
"prometheus-alerting-rules.yml",
|
|
111
|
+
"prometheus-alerting-templates.yml",
|
|
112
|
+
"alertmanager-simple.yml",
|
|
113
|
+
"alertmanager-route-labels.yml",
|
|
114
|
+
];
|
|
@@ -0,0 +1,42 @@
|
|
|
1
|
+
# Every rule file field, multi-line expressions (block scalars), a PromQL
|
|
2
|
+
# raw string in backticks, Go templates in annotations and a label value
|
|
3
|
+
# written as a number.
|
|
4
|
+
groups:
|
|
5
|
+
- name: node
|
|
6
|
+
interval: 1m
|
|
7
|
+
query_offset: 30s
|
|
8
|
+
limit: 100
|
|
9
|
+
labels:
|
|
10
|
+
team: platform
|
|
11
|
+
tier: 1
|
|
12
|
+
rules:
|
|
13
|
+
- record: instance:node_cpu_utilisation:rate5m
|
|
14
|
+
expr: |
|
|
15
|
+
1 - avg without (cpu, mode) (
|
|
16
|
+
rate(node_cpu_seconds_total{mode="idle"}[5m])
|
|
17
|
+
)
|
|
18
|
+
- record: instance:node_host:label
|
|
19
|
+
expr: label_replace(up{job="node"}, "host", "$1", "instance", `(.*):.*`)
|
|
20
|
+
labels:
|
|
21
|
+
source: node-exporter
|
|
22
|
+
- alert: NodeHighCpu
|
|
23
|
+
expr: instance:node_cpu_utilisation:rate5m > 0.9
|
|
24
|
+
for: 15m
|
|
25
|
+
keep_firing_for: 5m
|
|
26
|
+
labels:
|
|
27
|
+
severity: ticket
|
|
28
|
+
annotations:
|
|
29
|
+
summary: "{{ $labels.instance }} CPU above 90%"
|
|
30
|
+
description: >-
|
|
31
|
+
CPU on {{ $labels.instance }} has been above 90% for 15 minutes
|
|
32
|
+
(currently {{ $value | humanizePercentage }}).
|
|
33
|
+
- name: blackbox
|
|
34
|
+
rules:
|
|
35
|
+
- alert: ProbeFailing
|
|
36
|
+
expr: probe_success == 0
|
|
37
|
+
for: 2m
|
|
38
|
+
labels:
|
|
39
|
+
severity: page
|
|
40
|
+
annotations:
|
|
41
|
+
summary: "{{ $labels.instance }} is not answering"
|
|
42
|
+
runbook_url: https://runbooks.example.com/probe
|
|
@@ -0,0 +1,65 @@
|
|
|
1
|
+
# Vendored from github.com/prometheus/alertmanager v0.34.1, doc/examples/route_labels.yml
|
|
2
|
+
# (https://github.com/prometheus/alertmanager/blob/v0.34.1/doc/examples/route_labels.yml). Apache-2.0.
|
|
3
|
+
# Unchanged below this header.
|
|
4
|
+
# Example showing route `labels`.
|
|
5
|
+
#
|
|
6
|
+
# Route labels are attached to a route, inherited by child routes, and may be
|
|
7
|
+
# overridden per route. They are rendered per alert group and exposed to
|
|
8
|
+
# notification templates via the `routeLabels` function (and in the
|
|
9
|
+
# `routeLabels` field of the /api/v2/alerts/groups API response).
|
|
10
|
+
#
|
|
11
|
+
# The pattern shown here: a `description` is composed once at the root from a
|
|
12
|
+
# `reason` sub-label. Each subtree overrides only `reason`, computing it from
|
|
13
|
+
# the labels that branch matched on. The shared description picks up the new
|
|
14
|
+
# reason automatically, so the per-branch routes never restate the surrounding
|
|
15
|
+
# wording.
|
|
16
|
+
|
|
17
|
+
templates:
|
|
18
|
+
- '/etc/alertmanager/template/*.tmpl'
|
|
19
|
+
|
|
20
|
+
route:
|
|
21
|
+
group_by: ['alertname']
|
|
22
|
+
receiver: default
|
|
23
|
+
# `description` is defined once and built from `reason`. Child routes only
|
|
24
|
+
# override `reason`; they inherit this description unchanged.
|
|
25
|
+
labels:
|
|
26
|
+
reason: '{{ .GroupLabels.alertname }}'
|
|
27
|
+
description: '{{ .GroupLabels.alertname }} firing ({{ routeLabels "reason" }})'
|
|
28
|
+
routes:
|
|
29
|
+
# Database alerts are grouped by the affected database, so the reason can be
|
|
30
|
+
# computed from that branch's grouping label.
|
|
31
|
+
- matchers:
|
|
32
|
+
- service="database"
|
|
33
|
+
receiver: dba
|
|
34
|
+
group_by: ['alertname', 'database']
|
|
35
|
+
labels:
|
|
36
|
+
reason: 'database {{ .GroupLabels.database }}'
|
|
37
|
+
|
|
38
|
+
# Network link alerts match on both endpoints; the reason names the link.
|
|
39
|
+
- matchers:
|
|
40
|
+
- link_a_device=~".+"
|
|
41
|
+
- link_z_device=~".+"
|
|
42
|
+
receiver: network
|
|
43
|
+
group_by: ['alertname', 'link_a_device', 'link_z_device']
|
|
44
|
+
labels:
|
|
45
|
+
reason: '{{ .GroupLabels.link_a_device }} <-> {{ .GroupLabels.link_z_device }}'
|
|
46
|
+
|
|
47
|
+
receivers:
|
|
48
|
+
# The webhook payload includes the rendered route labels under "routeLabels".
|
|
49
|
+
- name: default
|
|
50
|
+
webhook_configs:
|
|
51
|
+
- url: 'http://127.0.0.1:5001/'
|
|
52
|
+
- name: dba
|
|
53
|
+
webhook_configs:
|
|
54
|
+
- url: 'http://127.0.0.1:5001/'
|
|
55
|
+
- name: network
|
|
56
|
+
webhook_configs:
|
|
57
|
+
- url: 'http://127.0.0.1:5001/'
|
|
58
|
+
|
|
59
|
+
# A notification template just renders the shared description:
|
|
60
|
+
#
|
|
61
|
+
# {{ define "route_labels.text" }}{{ routeLabels "description" }}{{ end }}
|
|
62
|
+
#
|
|
63
|
+
# For a generic alert this renders e.g. "HighLatency firing (HighLatency)";
|
|
64
|
+
# for the database branch "DiskFull firing (database orders)"; for the link
|
|
65
|
+
# branch "LinkDown firing (switch-a <-> switch-b)" -- only `reason` changed.
|
|
@@ -0,0 +1,130 @@
|
|
|
1
|
+
# Vendored from github.com/prometheus/alertmanager v0.34.1, doc/examples/simple.yml
|
|
2
|
+
# (https://github.com/prometheus/alertmanager/blob/v0.34.1/doc/examples/simple.yml). Apache-2.0.
|
|
3
|
+
# Unchanged below this header.
|
|
4
|
+
global:
|
|
5
|
+
# The smarthost and SMTP sender used for mail notifications.
|
|
6
|
+
smtp_smarthost: 'localhost:25'
|
|
7
|
+
smtp_from: 'alertmanager@example.org'
|
|
8
|
+
smtp_auth_username: 'alertmanager'
|
|
9
|
+
smtp_auth_password: 'password'
|
|
10
|
+
|
|
11
|
+
# The directory from which notification templates are read.
|
|
12
|
+
templates:
|
|
13
|
+
- '/etc/alertmanager/template/*.tmpl'
|
|
14
|
+
|
|
15
|
+
# The root route on which each incoming alert enters.
|
|
16
|
+
route:
|
|
17
|
+
# The labels by which incoming alerts are grouped together. For example,
|
|
18
|
+
# multiple alerts coming in for cluster=A and alertname=LatencyHigh would
|
|
19
|
+
# be batched into a single group.
|
|
20
|
+
#
|
|
21
|
+
# To aggregate by all possible labels use '...' as the sole label name.
|
|
22
|
+
# This effectively disables aggregation entirely, passing through all
|
|
23
|
+
# alerts as-is. This is unlikely to be what you want, unless you have
|
|
24
|
+
# a very low alert volume or your upstream notification system performs
|
|
25
|
+
# its own grouping. Example: group_by: [...]
|
|
26
|
+
group_by: ['alertname', 'cluster', 'service']
|
|
27
|
+
|
|
28
|
+
# When a new group of alerts is created by an incoming alert, wait at
|
|
29
|
+
# least 'group_wait' to send the initial notification.
|
|
30
|
+
# This way ensures that you get multiple alerts for the same group that start
|
|
31
|
+
# firing shortly after another are batched together on the first
|
|
32
|
+
# notification.
|
|
33
|
+
group_wait: 30s
|
|
34
|
+
|
|
35
|
+
# When the first notification was sent, wait 'group_interval' to send a batch
|
|
36
|
+
# of new alerts that started firing for that group.
|
|
37
|
+
group_interval: 5m
|
|
38
|
+
|
|
39
|
+
# If an alert has successfully been sent, wait 'repeat_interval' to
|
|
40
|
+
# resend them.
|
|
41
|
+
repeat_interval: 3h
|
|
42
|
+
|
|
43
|
+
# A default receiver
|
|
44
|
+
receiver: team-X-mails
|
|
45
|
+
|
|
46
|
+
# All the above attributes are inherited by all child routes and can
|
|
47
|
+
# overwritten on each.
|
|
48
|
+
|
|
49
|
+
# The child route trees.
|
|
50
|
+
routes:
|
|
51
|
+
# This routes performs a regular expression match on alert labels to
|
|
52
|
+
# catch alerts that are related to a list of services.
|
|
53
|
+
- matchers:
|
|
54
|
+
- service=~"foo1|foo2|baz"
|
|
55
|
+
receiver: team-X-mails
|
|
56
|
+
# The service has a sub-route for critical alerts, any alerts
|
|
57
|
+
# that do not match, i.e. severity != critical, fall-back to the
|
|
58
|
+
# parent node and are sent to 'team-X-mails'
|
|
59
|
+
routes:
|
|
60
|
+
- matchers:
|
|
61
|
+
- severity="critical"
|
|
62
|
+
receiver: team-X-pager
|
|
63
|
+
- matchers:
|
|
64
|
+
- service="files"
|
|
65
|
+
receiver: team-Y-mails
|
|
66
|
+
|
|
67
|
+
routes:
|
|
68
|
+
- matchers:
|
|
69
|
+
- severity="critical"
|
|
70
|
+
receiver: team-Y-pager
|
|
71
|
+
|
|
72
|
+
# This route handles all alerts coming from a database service. If there's
|
|
73
|
+
# no team to handle it, it defaults to the DB team.
|
|
74
|
+
- matchers:
|
|
75
|
+
- service="database"
|
|
76
|
+
receiver: team-DB-pager
|
|
77
|
+
# Also group alerts by affected database.
|
|
78
|
+
group_by: [alertname, cluster, database]
|
|
79
|
+
routes:
|
|
80
|
+
- matchers:
|
|
81
|
+
- owner="team-X"
|
|
82
|
+
receiver: team-X-pager
|
|
83
|
+
continue: true
|
|
84
|
+
- matchers:
|
|
85
|
+
- owner="team-Y"
|
|
86
|
+
receiver: team-Y-pager
|
|
87
|
+
|
|
88
|
+
|
|
89
|
+
# Inhibition rules allow to mute a set of alerts given that another alert is
|
|
90
|
+
# firing.
|
|
91
|
+
# We use this to mute any warning-level notifications if the same alert is
|
|
92
|
+
# already critical.
|
|
93
|
+
inhibit_rules:
|
|
94
|
+
- source_matchers: [severity="critical"]
|
|
95
|
+
target_matchers: [severity="warning"]
|
|
96
|
+
# Apply inhibition if the alertname is the same.
|
|
97
|
+
# CAUTION:
|
|
98
|
+
# If all label names listed in `equal` are missing
|
|
99
|
+
# from both the source and target alerts,
|
|
100
|
+
# the inhibition rule will apply!
|
|
101
|
+
equal: [alertname, cluster, service]
|
|
102
|
+
|
|
103
|
+
|
|
104
|
+
receivers:
|
|
105
|
+
- name: 'team-X-mails'
|
|
106
|
+
email_configs:
|
|
107
|
+
- to: 'team-X+alerts@example.org'
|
|
108
|
+
|
|
109
|
+
- name: 'team-X-pager'
|
|
110
|
+
email_configs:
|
|
111
|
+
- to: 'team-X+alerts-critical@example.org'
|
|
112
|
+
pagerduty_configs:
|
|
113
|
+
- service_key: <team-X-key>
|
|
114
|
+
|
|
115
|
+
- name: 'team-Y-mails'
|
|
116
|
+
email_configs:
|
|
117
|
+
- to: 'team-Y+alerts@example.org'
|
|
118
|
+
|
|
119
|
+
- name: 'team-Y-pager'
|
|
120
|
+
pagerduty_configs:
|
|
121
|
+
- service_key: <team-Y-key>
|
|
122
|
+
|
|
123
|
+
- name: 'team-DB-pager'
|
|
124
|
+
pagerduty_configs:
|
|
125
|
+
- service_key: <team-DB-key>
|
|
126
|
+
|
|
127
|
+
tracing:
|
|
128
|
+
endpoint: localhost:4317
|
|
129
|
+
insecure: true
|
|
130
|
+
sampling_fraction: 1.0
|
|
@@ -0,0 +1,15 @@
|
|
|
1
|
+
# From github.com/prometheus/prometheus v3.15.0, docs/configuration/alerting_rules.md,
|
|
2
|
+
# "Defining alerting rules" (https://github.com/prometheus/prometheus/blob/v3.15.0/docs/configuration/alerting_rules.md). Apache-2.0.
|
|
3
|
+
groups:
|
|
4
|
+
- name: example
|
|
5
|
+
labels:
|
|
6
|
+
team: myteam
|
|
7
|
+
rules:
|
|
8
|
+
- alert: HighRequestLatency
|
|
9
|
+
expr: job:request_latency_seconds:mean5m{job="myjob"} > 0.5
|
|
10
|
+
for: 10m
|
|
11
|
+
keep_firing_for: 5m
|
|
12
|
+
labels:
|
|
13
|
+
severity: page
|
|
14
|
+
annotations:
|
|
15
|
+
summary: High request latency
|
|
@@ -0,0 +1,23 @@
|
|
|
1
|
+
# From github.com/prometheus/prometheus v3.15.0, docs/configuration/alerting_rules.md,
|
|
2
|
+
# "Templating" example (https://github.com/prometheus/prometheus/blob/v3.15.0/docs/configuration/alerting_rules.md). Apache-2.0.
|
|
3
|
+
groups:
|
|
4
|
+
- name: example
|
|
5
|
+
rules:
|
|
6
|
+
|
|
7
|
+
# Alert for any instance that is unreachable for >5 minutes.
|
|
8
|
+
- alert: InstanceDown
|
|
9
|
+
expr: up == 0
|
|
10
|
+
for: 5m
|
|
11
|
+
labels:
|
|
12
|
+
severity: page
|
|
13
|
+
annotations:
|
|
14
|
+
summary: "Instance {{ $labels.instance }} down"
|
|
15
|
+
description: "{{ $labels.instance }} of job {{ $labels.job }} has been down for more than 5 minutes."
|
|
16
|
+
|
|
17
|
+
# Alert for any instance that has a median request latency >1s.
|
|
18
|
+
- alert: APIHighRequestLatency
|
|
19
|
+
expr: api_http_request_latencies_second{quantile="0.5"} > 1
|
|
20
|
+
for: 10m
|
|
21
|
+
annotations:
|
|
22
|
+
summary: "High request latency on {{ $labels.instance }}"
|
|
23
|
+
description: "{{ $labels.instance }} has a median request latency above 1s (current value: {{ $value }}s)"
|
|
@@ -0,0 +1,7 @@
|
|
|
1
|
+
# From github.com/prometheus/prometheus v3.15.0, docs/configuration/recording_rules.md,
|
|
2
|
+
# "A simple example rules file" (https://github.com/prometheus/prometheus/blob/v3.15.0/docs/configuration/recording_rules.md). Apache-2.0.
|
|
3
|
+
groups:
|
|
4
|
+
- name: example
|
|
5
|
+
rules:
|
|
6
|
+
- record: code:prometheus_http_requests_total:sum
|
|
7
|
+
expr: sum by (code) (prometheus_http_requests_total)
|