@intentius/chant-lexicon-prometheus 0.96.0 → 0.98.0

This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
Files changed (53) hide show
  1. package/README.md +5 -0
  2. package/dist/alertmanager.d.ts +4 -2
  3. package/dist/alertmanager.d.ts.map +1 -1
  4. package/dist/build.d.ts.map +1 -1
  5. package/dist/codegen/docs.d.ts.map +1 -1
  6. package/dist/import/generator.d.ts +35 -0
  7. package/dist/import/generator.d.ts.map +1 -0
  8. package/dist/import/parser.d.ts +52 -0
  9. package/dist/import/parser.d.ts.map +1 -0
  10. package/dist/import/slo.d.ts +15 -0
  11. package/dist/import/slo.d.ts.map +1 -0
  12. package/dist/integrity.json +7 -7
  13. package/dist/lint/rules/literal-credential.d.ts.map +1 -1
  14. package/dist/lint/rules/prom-ast.d.ts +8 -0
  15. package/dist/lint/rules/prom-ast.d.ts.map +1 -1
  16. package/dist/lint/rules/promql-literal.d.ts.map +1 -1
  17. package/dist/manifest.json +1 -1
  18. package/dist/model.d.ts +21 -0
  19. package/dist/model.d.ts.map +1 -1
  20. package/dist/plugin.d.ts.map +1 -1
  21. package/dist/rules/literal-credential.ts +17 -11
  22. package/dist/rules/prom-ast.ts +36 -0
  23. package/dist/rules/promql-literal.ts +15 -1
  24. package/dist/skills/chant-prometheus-alertmanager.md +4 -0
  25. package/dist/skills/chant-prometheus.md +11 -0
  26. package/package.json +2 -2
  27. package/src/alertmanager.ts +4 -1
  28. package/src/build.ts +2 -0
  29. package/src/codegen/docs.ts +5 -1
  30. package/src/import/cli.test.ts +98 -0
  31. package/src/import/generated-types.e2e.test.ts +75 -0
  32. package/src/import/generator.test.ts +121 -0
  33. package/src/import/generator.ts +570 -0
  34. package/src/import/parser.test.ts +120 -0
  35. package/src/import/parser.ts +348 -0
  36. package/src/import/roundtrip.test.ts +321 -0
  37. package/src/import/slo.ts +175 -0
  38. package/src/import/testdata/alertmanager-full.yml +89 -0
  39. package/src/import/testdata/fixtures.ts +114 -0
  40. package/src/import/testdata/rules-full.yml +42 -0
  41. package/src/import/testdata/upstream/alertmanager-route-labels.yml +65 -0
  42. package/src/import/testdata/upstream/alertmanager-simple.yml +130 -0
  43. package/src/import/testdata/upstream/prometheus-alerting-rules.yml +15 -0
  44. package/src/import/testdata/upstream/prometheus-alerting-templates.yml +23 -0
  45. package/src/import/testdata/upstream/prometheus-recording-rules.yml +7 -0
  46. package/src/lint/rules/literal-credential.ts +17 -11
  47. package/src/lint/rules/prom-ast.ts +36 -0
  48. package/src/lint/rules/promql-literal.ts +15 -1
  49. package/src/lint/rules/rules.test.ts +28 -0
  50. package/src/model.ts +22 -0
  51. package/src/plugin.ts +10 -0
  52. package/src/skills/chant-prometheus-alertmanager.md +4 -0
  53. package/src/skills/chant-prometheus.md +11 -0
@@ -0,0 +1,175 @@
1
+ /**
2
+ * Recognise a rule group that `Slo()` built, so the importer can write the
3
+ * `Slo` declaration instead of its forty-odd lines of rules.
4
+ *
5
+ * The group's name, recorded series, objective and burn-rate alerts give
6
+ * back candidate props; the candidate is accepted only when `Slo()` builds
7
+ * it to the same group, rule for rule. Anything else (a hand-edited
8
+ * threshold, an extra rule, a different description) fails that check and
9
+ * the group is imported as a plain `RuleGroup`.
10
+ */
11
+
12
+ import { Slo, DEFAULT_BURN_RATES, SLO_WINDOW_PLACEHOLDER, type SloAlertTier, type SloProps, type BurnRateWindow } from "../composites/slo";
13
+ import { ruleGroupConfig } from "../rules";
14
+ import { durationMs } from "../duration";
15
+ import { isAlertingRuleConfig, isRecordingRuleConfig, type AlertingRuleConfig, type LabelSet, type RuleGroupConfig } from "../model";
16
+
17
+ const RATIO_PREFIX = "slo:sli_error:ratio_rate";
18
+ const OBJECTIVE = "slo:objective:ratio";
19
+ const BUDGET = "slo:error_budget:remaining";
20
+ const DEFAULT_ALERT_NAME = "ErrorBudgetBurn";
21
+ const PAIR_LABELS = new Set(["slo", "severity", "long_window", "short_window"]);
22
+ const DESCRIPTION_TAIL = "The error ratio over both the last ";
23
+
24
+ /** Objects with their keys sorted, so two configs compare equal whatever order their keys were written in. */
25
+ function canonical(v: unknown): unknown {
26
+ if (Array.isArray(v)) return v.map(canonical);
27
+ if (typeof v === "object" && v !== null) {
28
+ return Object.fromEntries(
29
+ Object.keys(v)
30
+ .sort()
31
+ .map((k) => [k, canonical((v as Record<string, unknown>)[k])]),
32
+ );
33
+ }
34
+ return v;
35
+ }
36
+
37
+ function same(a: unknown, b: unknown): boolean {
38
+ return JSON.stringify(canonical(a)) === JSON.stringify(canonical(b));
39
+ }
40
+
41
+ /** The SLI expressions a recorded error ratio was built from, with the window put back as `{{window}}`. */
42
+ function sliFrom(expr: string, window: string): SloProps["sli"] | undefined {
43
+ const unwindow = (s: string) => s.split(`[${window}]`).join(`[${SLO_WINDOW_PLACEHOLDER}]`);
44
+ const goodHead = "1 - (\n (";
45
+ const goodTail = ")\n)";
46
+ if (expr.startsWith(goodHead) && expr.endsWith(goodTail)) {
47
+ const inner = expr.slice(goodHead.length, -goodTail.length);
48
+ const at = inner.indexOf(")\n /\n (");
49
+ if (at === -1) return undefined;
50
+ return { good: unwindow(inner.slice(0, at)), total: unwindow(inner.slice(at + ")\n /\n (".length)) };
51
+ }
52
+ if (expr.startsWith("(") && expr.endsWith(")")) {
53
+ const inner = expr.slice(1, -1);
54
+ const at = inner.indexOf(")\n/\n(");
55
+ if (at === -1) return undefined;
56
+ return { errors: unwindow(inner.slice(0, at)), total: unwindow(inner.slice(at + ")\n/\n(".length)) };
57
+ }
58
+ return undefined;
59
+ }
60
+
61
+ function extras(set: LabelSet | undefined, skip: (k: string) => boolean): LabelSet | undefined {
62
+ const out: LabelSet = {};
63
+ for (const [k, v] of Object.entries(set ?? {})) if (!skip(k)) out[k] = v;
64
+ return Object.keys(out).length > 0 ? out : undefined;
65
+ }
66
+
67
+ function isDefaultPairs(tier: "page" | "ticket", pairs: BurnRateWindow[], factors: number[], windowMs: (w: string) => number, sloWindow: string): boolean {
68
+ const defaults = DEFAULT_BURN_RATES[tier];
69
+ if (defaults.length !== pairs.length) return false;
70
+ return defaults.every((d, i) => {
71
+ const p = pairs[i];
72
+ const factor = Number(((d.budgetConsumed! * windowMs(sloWindow)) / windowMs(d.long)).toPrecision(6));
73
+ return p.long === d.long && p.short === d.short && factor === factors[i];
74
+ });
75
+ }
76
+
77
+ /** The `Slo` props that build exactly `group`, or `undefined` when no `Slo()` call does. */
78
+ export function recognizeSlo(group: RuleGroupConfig): SloProps | undefined {
79
+ if (!group.name.startsWith("slo-") || group.query_offset !== undefined || group.limit !== undefined) return undefined;
80
+ const name = group.name.slice("slo-".length);
81
+ const records = group.rules.filter(isRecordingRuleConfig);
82
+ const alerts = group.rules.filter(isAlertingRuleConfig) as AlertingRuleConfig[];
83
+ if (records.length + alerts.length !== group.rules.length) return undefined;
84
+
85
+ const objectiveAt = records.findIndex((r) => r.record === OBJECTIVE);
86
+ if (objectiveAt < 1 || records[objectiveAt + 1]?.record !== BUDGET) return undefined;
87
+ const objective = Number(/^vector\((.+)\)$/.exec(records[objectiveAt].expr)?.[1]);
88
+ if (!Number.isFinite(objective)) return undefined;
89
+ const windowRecord = records[objectiveAt - 1].record;
90
+ if (!windowRecord.startsWith(RATIO_PREFIX)) return undefined;
91
+ const window = windowRecord.slice(RATIO_PREFIX.length);
92
+
93
+ // Candidate SLIs: from each alert window's ratio, or from the whole-window
94
+ // ratio when there are no alerts.
95
+ const ratioRecords = records.slice(0, objectiveAt - 1);
96
+ const sources = ratioRecords.length > 0 ? ratioRecords : [records[objectiveAt - 1]];
97
+ const slis = sources
98
+ .map((r) => (r.record.startsWith(RATIO_PREFIX) ? sliFrom(r.expr, r.record.slice(RATIO_PREFIX.length)) : undefined))
99
+ .filter((s): s is SloProps["sli"] => s !== undefined);
100
+ if (slis.length === 0) return undefined;
101
+
102
+ // Alert tiers: `page` and `ticket` by severity, or custom severities in the order they appear.
103
+ const alertName = alerts[0]?.alert ?? DEFAULT_ALERT_NAME;
104
+ const bySeverity = new Map<string, AlertingRuleConfig[]>();
105
+ for (const a of alerts) {
106
+ const severity = a.labels?.severity;
107
+ if (severity === undefined || a.alert !== alertName) return undefined;
108
+ bySeverity.set(severity, [...(bySeverity.get(severity) ?? []), a]);
109
+ }
110
+ if (bySeverity.size > 2) return undefined;
111
+ const severities = [...bySeverity.keys()];
112
+ const tierOf = new Map<string, "page" | "ticket">();
113
+ for (const s of severities) if (s === "page" || s === "ticket") tierOf.set(s, s);
114
+ for (const s of severities) {
115
+ if (tierOf.has(s)) continue;
116
+ const free = (["page", "ticket"] as const).find((t) => ![...tierOf.values()].includes(t));
117
+ if (!free) return undefined;
118
+ tierOf.set(s, free);
119
+ }
120
+
121
+ let description: string | undefined;
122
+ const alerting: NonNullable<SloProps["alerting"]> = {};
123
+ for (const tier of ["page", "ticket"] as const) {
124
+ const severity = [...tierOf].find(([, t]) => t === tier)?.[0];
125
+ if (severity === undefined) {
126
+ alerting[tier] = false;
127
+ continue;
128
+ }
129
+ const tierAlerts = bySeverity.get(severity)!;
130
+ const first = tierAlerts[0];
131
+ const pairs: BurnRateWindow[] = [];
132
+ const factors: number[] = [];
133
+ for (const a of tierAlerts) {
134
+ const long = a.labels?.long_window;
135
+ const short = a.labels?.short_window;
136
+ const factor = Number(/> \(([^ ]+) \* /.exec(a.expr)?.[1]);
137
+ if (long === undefined || short === undefined || !Number.isFinite(factor)) return undefined;
138
+ pairs.push({ long, short, factor });
139
+ factors.push(factor);
140
+ }
141
+ const t: SloAlertTier = {};
142
+ t.burnRates = isDefaultPairs(tier, pairs, factors, (w) => durationMs(w) ?? 0, window) ? "default" : pairs;
143
+ if (severity !== tier) t.severity = severity;
144
+ if (first.for !== undefined) t.for = first.for;
145
+ const labels = extras(first.labels, (k) => PAIR_LABELS.has(k));
146
+ if (labels) t.labels = labels;
147
+ const annotations = extras(first.annotations, (k) => k === "summary" || k === "description");
148
+ if (annotations) t.annotations = annotations;
149
+ const desc = first.annotations?.description ?? "";
150
+ const at = desc.indexOf(DESCRIPTION_TAIL);
151
+ if (at > 0 && description === undefined) description = desc.slice(0, at).trimEnd();
152
+ const onlyDefault = t.burnRates === "default" && Object.keys(t).length === 1;
153
+ if (!onlyDefault) alerting[tier] = t;
154
+ }
155
+ if (alertName !== DEFAULT_ALERT_NAME) alerting.alertName = alertName;
156
+
157
+ for (const sli of slis) {
158
+ const props: SloProps = {
159
+ name,
160
+ objective,
161
+ window,
162
+ ...(description ? { description } : {}),
163
+ sli,
164
+ ...(Object.keys(alerting).length > 0 ? { alerting } : {}),
165
+ ...(group.labels ? { labels: group.labels } : {}),
166
+ ...(group.interval !== undefined ? { interval: group.interval } : {}),
167
+ };
168
+ try {
169
+ if (same(ruleGroupConfig(Slo(props).rules), ruleGroupConfig(group))) return props;
170
+ } catch {
171
+ // Not buildable from these props; try the next candidate.
172
+ }
173
+ }
174
+ return undefined;
175
+ }
@@ -0,0 +1,89 @@
1
+ # Deprecated spellings (match, match_re, source_match, the top-level
2
+ # mute_time_intervals), integrations this lexicon does not type (opsgenie,
3
+ # msteams), global fields outside the typed set, *_file credentials, Go
4
+ # templates, a YAML anchor with a merge key, and one literal credential.
5
+ global:
6
+ resolve_timeout: 5m
7
+ smtp_smarthost: smtp.example.com:587
8
+ smtp_from: alertmanager@example.com
9
+ smtp_auth_username: alertmanager
10
+ smtp_auth_password_file: /etc/alertmanager/secrets/smtp-password
11
+ slack_api_url_file: /etc/alertmanager/secrets/slack-url
12
+ opsgenie_api_key_file: /etc/alertmanager/secrets/opsgenie-key
13
+ http_config:
14
+ follow_redirects: true
15
+ templates:
16
+ - /etc/alertmanager/templates/*.tmpl
17
+ route:
18
+ receiver: default
19
+ group_by: [alertname, cluster]
20
+ group_wait: 30s
21
+ labels:
22
+ summary: '{{ .GroupLabels.alertname }} in {{ .GroupLabels.cluster }}'
23
+ routes:
24
+ - match:
25
+ severity: page
26
+ match_re:
27
+ service: ^(api|web)$
28
+ receiver: oncall
29
+ continue: true
30
+ - matchers: ['team="db"']
31
+ receiver: db-chat
32
+ mute_time_intervals: [weekends]
33
+ active_time_intervals: [business-hours]
34
+ - match:
35
+ alertname: Watchdog
36
+ receiver: heartbeat
37
+ repeat_interval: 1m
38
+ inhibit_rules:
39
+ - name: page-mutes-ticket
40
+ source_match:
41
+ severity: page
42
+ target_matchers: ['severity="ticket"']
43
+ equal: [alertname, cluster]
44
+ receivers:
45
+ - name: default
46
+ webhook_configs:
47
+ - url_file: /etc/alertmanager/secrets/default-webhook-url
48
+ max_alerts: 10
49
+ - name: oncall
50
+ pagerduty_configs:
51
+ - routing_key_file: /etc/alertmanager/secrets/pagerduty-key
52
+ description: '{{ template "pagerduty.default.description" . }}'
53
+ details:
54
+ firing: '{{ .Alerts.Firing | len }}'
55
+ opsgenie_configs:
56
+ - api_key_file: /etc/alertmanager/secrets/opsgenie-key
57
+ priority: P1
58
+ - name: db-chat
59
+ slack_configs:
60
+ - &slack
61
+ channel: '#db-alerts'
62
+ title: '{{ .CommonLabels.alertname }}'
63
+ text: >-
64
+ {{ range .Alerts }}{{ .Annotations.summary }}
65
+ {{ end }}
66
+ send_resolved: true
67
+ - <<: *slack
68
+ channel: '#db-oncall'
69
+ msteams_configs:
70
+ - webhook_url_file: /etc/alertmanager/secrets/teams-url
71
+ - name: heartbeat
72
+ email_configs:
73
+ - to: heartbeat@example.com
74
+ auth_username: heartbeat
75
+ auth_password: hunter2
76
+ headers:
77
+ Subject: '[heartbeat] {{ .CommonLabels.alertname }}'
78
+ time_intervals:
79
+ - name: business-hours
80
+ time_intervals:
81
+ - weekdays: ['monday:friday']
82
+ times:
83
+ - start_time: '09:00'
84
+ end_time: '17:00'
85
+ location: Europe/Berlin
86
+ mute_time_intervals:
87
+ - name: weekends
88
+ time_intervals:
89
+ - weekdays: [saturday, sunday]
@@ -0,0 +1,114 @@
1
+ /**
2
+ * Fixtures shared by the import round-trip tests (roundtrip.test.ts,
3
+ * cli.test.ts) and the type-check of the generated source
4
+ * (generated-types.e2e.test.ts).
5
+ */
6
+ import { readdirSync, readFileSync, statSync } from "fs";
7
+ import { join, resolve } from "path";
8
+ import { build } from "@intentius/chant/build";
9
+ import type { Serializer, SerializerResult } from "@intentius/chant/serializer";
10
+ import { k8sSerializer } from "@intentius/chant-lexicon-k8s";
11
+ import { prometheusSerializer, ALERTMANAGER_FILE } from "../../serializer";
12
+ import { ruleFileYaml } from "../../build";
13
+ import { Slo } from "../../composites/slo";
14
+
15
+ export const pkgDir = resolve(import.meta.dirname, "../../..");
16
+ export const repoRoot = resolve(pkgDir, "../..");
17
+ export const read = (...p: string[]) => readFileSync(join(import.meta.dirname, ...p), "utf-8");
18
+
19
+ /** One file a build wrote: the rule file or `alertmanager.yml`. */
20
+ export interface BuiltFile {
21
+ name: string;
22
+ yaml: string;
23
+ }
24
+
25
+ /** The rule file and `alertmanager.yml` of one build output, whichever it has. */
26
+ export function builtFiles(out: string | SerializerResult | undefined): { rules?: string; alertmanager?: string } {
27
+ if (out === undefined || out === "") return {};
28
+ if (typeof out === "string") return out.startsWith("groups:") ? { rules: out } : { alertmanager: out };
29
+ const primaryIsRules = out.primary.startsWith("groups:");
30
+ return {
31
+ ...(primaryIsRules ? { rules: out.primary } : { alertmanager: out.primary }),
32
+ ...(out.files?.[ALERTMANAGER_FILE] !== undefined ? { alertmanager: out.files[ALERTMANAGER_FILE] } : {}),
33
+ };
34
+ }
35
+
36
+ /** Every file each prometheus example builds, named `<example>/rules.yml` or `<example>/alertmanager.yml`. */
37
+ export async function exampleOutputs(): Promise<BuiltFile[]> {
38
+ const examplesDir = join(pkgDir, "examples");
39
+ const out: BuiltFile[] = [];
40
+ for (const name of readdirSync(examplesDir).sort()) {
41
+ const srcDir = join(examplesDir, name, "src");
42
+ try {
43
+ if (!statSync(srcDir).isDirectory()) continue;
44
+ } catch {
45
+ continue;
46
+ }
47
+ // k3d-stack declares its Kubernetes workloads beside the rules.
48
+ const serializers: Serializer[] = name === "k3d-stack" ? [k8sSerializer, prometheusSerializer] : [prometheusSerializer];
49
+ const result = await build(srcDir, serializers);
50
+ if (result.errors.length > 0) throw new Error(`${name}: ${result.errors.map(String).join("; ")}`);
51
+ const files = builtFiles(result.outputs.get("prometheus"));
52
+ if (files.rules !== undefined) out.push({ name: `${name}/rules.yml`, yaml: files.rules });
53
+ if (files.alertmanager !== undefined) out.push({ name: `${name}/alertmanager.yml`, yaml: files.alertmanager });
54
+ }
55
+ return out;
56
+ }
57
+
58
+ const calls = "traces_span_metrics_calls_total";
59
+
60
+ /** Rule files `Slo()` builds, over the props that change its shape. */
61
+ export function sloOutputs(): BuiltFile[] {
62
+ const good = {
63
+ good: `sum(rate(${calls}{span_name="checkout",status_code!="STATUS_CODE_ERROR"}[{{window}}]))`,
64
+ total: `sum(rate(${calls}{span_name="checkout"}[{{window}}]))`,
65
+ };
66
+ const errors = {
67
+ errors: `sum(rate(http_requests_total{code=~"5.."}[{{window}}]))`,
68
+ total: "sum(rate(http_requests_total[{{window}}]))",
69
+ };
70
+ const slos = [
71
+ Slo({ name: "defaults", objective: 0.999, window: "30d", sli: errors }),
72
+ Slo({
73
+ name: "order-ack",
74
+ objective: 0.995,
75
+ window: "28d",
76
+ description: "Orders are acknowledged without an error span.",
77
+ sli: good,
78
+ alerting: {
79
+ page: { burnRates: "default", annotations: { runbook_url: "https://runbooks.example.com/order-ack" } },
80
+ ticket: { burnRates: "default", for: "15m", labels: { team: "orders" } },
81
+ },
82
+ labels: { team: "orders" },
83
+ interval: "1m",
84
+ }),
85
+ Slo({
86
+ name: "custom",
87
+ objective: 0.99,
88
+ window: "7d",
89
+ sli: errors,
90
+ alerting: {
91
+ alertName: "LatencyBudgetBurn",
92
+ page: {
93
+ severity: "critical",
94
+ burnRates: [
95
+ { long: "1h", short: "5m", factor: 10 },
96
+ { long: "6h", short: "30m", budgetConsumed: 0.1 },
97
+ ],
98
+ },
99
+ ticket: false,
100
+ },
101
+ }),
102
+ Slo({ name: "no-alerts", objective: 0.95, window: "1w", sli: good, alerting: { page: false, ticket: false } }),
103
+ ];
104
+ return slos.map((s) => ({ name: `Slo ${s.rules.props.name}`, yaml: ruleFileYaml([s.rules]) }));
105
+ }
106
+
107
+ /** The vendored upstream samples, each with the source it came from in its header. */
108
+ export const UPSTREAM = [
109
+ "prometheus-recording-rules.yml",
110
+ "prometheus-alerting-rules.yml",
111
+ "prometheus-alerting-templates.yml",
112
+ "alertmanager-simple.yml",
113
+ "alertmanager-route-labels.yml",
114
+ ];
@@ -0,0 +1,42 @@
1
+ # Every rule file field, multi-line expressions (block scalars), a PromQL
2
+ # raw string in backticks, Go templates in annotations and a label value
3
+ # written as a number.
4
+ groups:
5
+ - name: node
6
+ interval: 1m
7
+ query_offset: 30s
8
+ limit: 100
9
+ labels:
10
+ team: platform
11
+ tier: 1
12
+ rules:
13
+ - record: instance:node_cpu_utilisation:rate5m
14
+ expr: |
15
+ 1 - avg without (cpu, mode) (
16
+ rate(node_cpu_seconds_total{mode="idle"}[5m])
17
+ )
18
+ - record: instance:node_host:label
19
+ expr: label_replace(up{job="node"}, "host", "$1", "instance", `(.*):.*`)
20
+ labels:
21
+ source: node-exporter
22
+ - alert: NodeHighCpu
23
+ expr: instance:node_cpu_utilisation:rate5m > 0.9
24
+ for: 15m
25
+ keep_firing_for: 5m
26
+ labels:
27
+ severity: ticket
28
+ annotations:
29
+ summary: "{{ $labels.instance }} CPU above 90%"
30
+ description: >-
31
+ CPU on {{ $labels.instance }} has been above 90% for 15 minutes
32
+ (currently {{ $value | humanizePercentage }}).
33
+ - name: blackbox
34
+ rules:
35
+ - alert: ProbeFailing
36
+ expr: probe_success == 0
37
+ for: 2m
38
+ labels:
39
+ severity: page
40
+ annotations:
41
+ summary: "{{ $labels.instance }} is not answering"
42
+ runbook_url: https://runbooks.example.com/probe
@@ -0,0 +1,65 @@
1
+ # Vendored from github.com/prometheus/alertmanager v0.34.1, doc/examples/route_labels.yml
2
+ # (https://github.com/prometheus/alertmanager/blob/v0.34.1/doc/examples/route_labels.yml). Apache-2.0.
3
+ # Unchanged below this header.
4
+ # Example showing route `labels`.
5
+ #
6
+ # Route labels are attached to a route, inherited by child routes, and may be
7
+ # overridden per route. They are rendered per alert group and exposed to
8
+ # notification templates via the `routeLabels` function (and in the
9
+ # `routeLabels` field of the /api/v2/alerts/groups API response).
10
+ #
11
+ # The pattern shown here: a `description` is composed once at the root from a
12
+ # `reason` sub-label. Each subtree overrides only `reason`, computing it from
13
+ # the labels that branch matched on. The shared description picks up the new
14
+ # reason automatically, so the per-branch routes never restate the surrounding
15
+ # wording.
16
+
17
+ templates:
18
+ - '/etc/alertmanager/template/*.tmpl'
19
+
20
+ route:
21
+ group_by: ['alertname']
22
+ receiver: default
23
+ # `description` is defined once and built from `reason`. Child routes only
24
+ # override `reason`; they inherit this description unchanged.
25
+ labels:
26
+ reason: '{{ .GroupLabels.alertname }}'
27
+ description: '{{ .GroupLabels.alertname }} firing ({{ routeLabels "reason" }})'
28
+ routes:
29
+ # Database alerts are grouped by the affected database, so the reason can be
30
+ # computed from that branch's grouping label.
31
+ - matchers:
32
+ - service="database"
33
+ receiver: dba
34
+ group_by: ['alertname', 'database']
35
+ labels:
36
+ reason: 'database {{ .GroupLabels.database }}'
37
+
38
+ # Network link alerts match on both endpoints; the reason names the link.
39
+ - matchers:
40
+ - link_a_device=~".+"
41
+ - link_z_device=~".+"
42
+ receiver: network
43
+ group_by: ['alertname', 'link_a_device', 'link_z_device']
44
+ labels:
45
+ reason: '{{ .GroupLabels.link_a_device }} <-> {{ .GroupLabels.link_z_device }}'
46
+
47
+ receivers:
48
+ # The webhook payload includes the rendered route labels under "routeLabels".
49
+ - name: default
50
+ webhook_configs:
51
+ - url: 'http://127.0.0.1:5001/'
52
+ - name: dba
53
+ webhook_configs:
54
+ - url: 'http://127.0.0.1:5001/'
55
+ - name: network
56
+ webhook_configs:
57
+ - url: 'http://127.0.0.1:5001/'
58
+
59
+ # A notification template just renders the shared description:
60
+ #
61
+ # {{ define "route_labels.text" }}{{ routeLabels "description" }}{{ end }}
62
+ #
63
+ # For a generic alert this renders e.g. "HighLatency firing (HighLatency)";
64
+ # for the database branch "DiskFull firing (database orders)"; for the link
65
+ # branch "LinkDown firing (switch-a <-> switch-b)" -- only `reason` changed.
@@ -0,0 +1,130 @@
1
+ # Vendored from github.com/prometheus/alertmanager v0.34.1, doc/examples/simple.yml
2
+ # (https://github.com/prometheus/alertmanager/blob/v0.34.1/doc/examples/simple.yml). Apache-2.0.
3
+ # Unchanged below this header.
4
+ global:
5
+ # The smarthost and SMTP sender used for mail notifications.
6
+ smtp_smarthost: 'localhost:25'
7
+ smtp_from: 'alertmanager@example.org'
8
+ smtp_auth_username: 'alertmanager'
9
+ smtp_auth_password: 'password'
10
+
11
+ # The directory from which notification templates are read.
12
+ templates:
13
+ - '/etc/alertmanager/template/*.tmpl'
14
+
15
+ # The root route on which each incoming alert enters.
16
+ route:
17
+ # The labels by which incoming alerts are grouped together. For example,
18
+ # multiple alerts coming in for cluster=A and alertname=LatencyHigh would
19
+ # be batched into a single group.
20
+ #
21
+ # To aggregate by all possible labels use '...' as the sole label name.
22
+ # This effectively disables aggregation entirely, passing through all
23
+ # alerts as-is. This is unlikely to be what you want, unless you have
24
+ # a very low alert volume or your upstream notification system performs
25
+ # its own grouping. Example: group_by: [...]
26
+ group_by: ['alertname', 'cluster', 'service']
27
+
28
+ # When a new group of alerts is created by an incoming alert, wait at
29
+ # least 'group_wait' to send the initial notification.
30
+ # This way ensures that you get multiple alerts for the same group that start
31
+ # firing shortly after another are batched together on the first
32
+ # notification.
33
+ group_wait: 30s
34
+
35
+ # When the first notification was sent, wait 'group_interval' to send a batch
36
+ # of new alerts that started firing for that group.
37
+ group_interval: 5m
38
+
39
+ # If an alert has successfully been sent, wait 'repeat_interval' to
40
+ # resend them.
41
+ repeat_interval: 3h
42
+
43
+ # A default receiver
44
+ receiver: team-X-mails
45
+
46
+ # All the above attributes are inherited by all child routes and can
47
+ # overwritten on each.
48
+
49
+ # The child route trees.
50
+ routes:
51
+ # This routes performs a regular expression match on alert labels to
52
+ # catch alerts that are related to a list of services.
53
+ - matchers:
54
+ - service=~"foo1|foo2|baz"
55
+ receiver: team-X-mails
56
+ # The service has a sub-route for critical alerts, any alerts
57
+ # that do not match, i.e. severity != critical, fall-back to the
58
+ # parent node and are sent to 'team-X-mails'
59
+ routes:
60
+ - matchers:
61
+ - severity="critical"
62
+ receiver: team-X-pager
63
+ - matchers:
64
+ - service="files"
65
+ receiver: team-Y-mails
66
+
67
+ routes:
68
+ - matchers:
69
+ - severity="critical"
70
+ receiver: team-Y-pager
71
+
72
+ # This route handles all alerts coming from a database service. If there's
73
+ # no team to handle it, it defaults to the DB team.
74
+ - matchers:
75
+ - service="database"
76
+ receiver: team-DB-pager
77
+ # Also group alerts by affected database.
78
+ group_by: [alertname, cluster, database]
79
+ routes:
80
+ - matchers:
81
+ - owner="team-X"
82
+ receiver: team-X-pager
83
+ continue: true
84
+ - matchers:
85
+ - owner="team-Y"
86
+ receiver: team-Y-pager
87
+
88
+
89
+ # Inhibition rules allow to mute a set of alerts given that another alert is
90
+ # firing.
91
+ # We use this to mute any warning-level notifications if the same alert is
92
+ # already critical.
93
+ inhibit_rules:
94
+ - source_matchers: [severity="critical"]
95
+ target_matchers: [severity="warning"]
96
+ # Apply inhibition if the alertname is the same.
97
+ # CAUTION:
98
+ # If all label names listed in `equal` are missing
99
+ # from both the source and target alerts,
100
+ # the inhibition rule will apply!
101
+ equal: [alertname, cluster, service]
102
+
103
+
104
+ receivers:
105
+ - name: 'team-X-mails'
106
+ email_configs:
107
+ - to: 'team-X+alerts@example.org'
108
+
109
+ - name: 'team-X-pager'
110
+ email_configs:
111
+ - to: 'team-X+alerts-critical@example.org'
112
+ pagerduty_configs:
113
+ - service_key: <team-X-key>
114
+
115
+ - name: 'team-Y-mails'
116
+ email_configs:
117
+ - to: 'team-Y+alerts@example.org'
118
+
119
+ - name: 'team-Y-pager'
120
+ pagerduty_configs:
121
+ - service_key: <team-Y-key>
122
+
123
+ - name: 'team-DB-pager'
124
+ pagerduty_configs:
125
+ - service_key: <team-DB-key>
126
+
127
+ tracing:
128
+ endpoint: localhost:4317
129
+ insecure: true
130
+ sampling_fraction: 1.0
@@ -0,0 +1,15 @@
1
+ # From github.com/prometheus/prometheus v3.15.0, docs/configuration/alerting_rules.md,
2
+ # "Defining alerting rules" (https://github.com/prometheus/prometheus/blob/v3.15.0/docs/configuration/alerting_rules.md). Apache-2.0.
3
+ groups:
4
+ - name: example
5
+ labels:
6
+ team: myteam
7
+ rules:
8
+ - alert: HighRequestLatency
9
+ expr: job:request_latency_seconds:mean5m{job="myjob"} > 0.5
10
+ for: 10m
11
+ keep_firing_for: 5m
12
+ labels:
13
+ severity: page
14
+ annotations:
15
+ summary: High request latency
@@ -0,0 +1,23 @@
1
+ # From github.com/prometheus/prometheus v3.15.0, docs/configuration/alerting_rules.md,
2
+ # "Templating" example (https://github.com/prometheus/prometheus/blob/v3.15.0/docs/configuration/alerting_rules.md). Apache-2.0.
3
+ groups:
4
+ - name: example
5
+ rules:
6
+
7
+ # Alert for any instance that is unreachable for >5 minutes.
8
+ - alert: InstanceDown
9
+ expr: up == 0
10
+ for: 5m
11
+ labels:
12
+ severity: page
13
+ annotations:
14
+ summary: "Instance {{ $labels.instance }} down"
15
+ description: "{{ $labels.instance }} of job {{ $labels.job }} has been down for more than 5 minutes."
16
+
17
+ # Alert for any instance that has a median request latency >1s.
18
+ - alert: APIHighRequestLatency
19
+ expr: api_http_request_latencies_second{quantile="0.5"} > 1
20
+ for: 10m
21
+ annotations:
22
+ summary: "High request latency on {{ $labels.instance }}"
23
+ description: "{{ $labels.instance }} has a median request latency above 1s (current value: {{ $value }}s)"
@@ -0,0 +1,7 @@
1
+ # From github.com/prometheus/prometheus v3.15.0, docs/configuration/recording_rules.md,
2
+ # "A simple example rules file" (https://github.com/prometheus/prometheus/blob/v3.15.0/docs/configuration/recording_rules.md). Apache-2.0.
3
+ groups:
4
+ - name: example
5
+ rules:
6
+ - record: code:prometheus_http_requests_total:sum
7
+ expr: sum by (code) (prometheus_http_requests_total)