@intentius/chant-lexicon-k8s 0.98.0 → 0.99.0

This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
@@ -0,0 +1,2905 @@
1
+ apiVersion: v1
2
+ kind: Namespace
3
+ metadata:
4
+ name: agents
5
+ labels:
6
+ app.kubernetes.io/name: agents
7
+ app.kubernetes.io/managed-by: chant
8
+ app.kubernetes.io/part-of: agent-observability
9
+ app.kubernetes.io/component: namespace
10
+
11
+ ---
12
+ apiVersion: v1
13
+ kind: Namespace
14
+ metadata:
15
+ name: observability
16
+ labels:
17
+ app.kubernetes.io/name: observability
18
+ app.kubernetes.io/managed-by: chant
19
+ app.kubernetes.io/part-of: agent-observability
20
+ app.kubernetes.io/component: namespace
21
+
22
+ ---
23
+ apiVersion: apps/v1
24
+ kind: DaemonSet
25
+ metadata:
26
+ name: otel-agent
27
+ namespace: observability
28
+ labels:
29
+ app.kubernetes.io/name: otel-agent
30
+ app.kubernetes.io/managed-by: chant
31
+ app.kubernetes.io/component: agent
32
+ annotations:
33
+ otel.chant.dev/role: agent
34
+ otel.chant.dev/config: otel-agent-config
35
+ otel.chant.dev/gateways: observability/otel-gateway=loadbalancing,observability/otel-gateway=service
36
+ spec:
37
+ selector:
38
+ matchLabels:
39
+ app.kubernetes.io/name: otel-agent
40
+ template:
41
+ metadata:
42
+ labels:
43
+ app.kubernetes.io/name: otel-agent
44
+ spec:
45
+ serviceAccountName: otel-agent-sa
46
+ containers:
47
+ - name: otel-agent
48
+ image: otel/opentelemetry-collector-contrib:0.130.0
49
+ args:
50
+ - --config=/etc/otel/config.yaml
51
+ ports:
52
+ - containerPort: 4317
53
+ name: otlp-grpc
54
+ - containerPort: 4318
55
+ name: otlp-http
56
+ - containerPort: 13133
57
+ name: health
58
+ resources:
59
+ requests:
60
+ cpu: '100m'
61
+ memory: '256Mi'
62
+ limits:
63
+ cpu: '500m'
64
+ memory: '512Mi'
65
+ volumeMounts:
66
+ - name: config
67
+ mountPath: /etc/otel
68
+ readOnly: true
69
+ securityContext:
70
+ runAsNonRoot: true
71
+ runAsUser: 10001
72
+ readOnlyRootFilesystem: true
73
+ allowPrivilegeEscalation: false
74
+ capabilities:
75
+ drop:
76
+ - ALL
77
+ imagePullPolicy: IfNotPresent
78
+ livenessProbe:
79
+ httpGet:
80
+ path: /
81
+ port: health
82
+ readinessProbe:
83
+ httpGet:
84
+ path: /
85
+ port: health
86
+ volumes:
87
+ - name: config
88
+ configMap:
89
+ name: otel-agent-config
90
+ tolerations:
91
+ - operator: Exists
92
+
93
+ ---
94
+ apiVersion: v1
95
+ kind: ServiceAccount
96
+ metadata:
97
+ name: otel-agent-sa
98
+ namespace: observability
99
+ labels:
100
+ app.kubernetes.io/name: otel-agent
101
+ app.kubernetes.io/managed-by: chant
102
+ app.kubernetes.io/component: agent
103
+
104
+ ---
105
+ apiVersion: rbac.authorization.k8s.io/v1
106
+ kind: ClusterRole
107
+ metadata:
108
+ name: otel-agent-role
109
+ labels:
110
+ app.kubernetes.io/name: otel-agent
111
+ app.kubernetes.io/managed-by: chant
112
+ app.kubernetes.io/component: rbac
113
+ rules:
114
+ - apiGroups:
115
+ - ''
116
+ resources:
117
+ - pods
118
+ - nodes
119
+ - endpoints
120
+ verbs:
121
+ - get
122
+ - list
123
+ - watch
124
+ - apiGroups:
125
+ - apps
126
+ resources:
127
+ - replicasets
128
+ verbs:
129
+ - get
130
+ - list
131
+ - watch
132
+ - apiGroups:
133
+ - batch
134
+ resources:
135
+ - jobs
136
+ verbs:
137
+ - get
138
+ - list
139
+ - watch
140
+ - apiGroups:
141
+ - ''
142
+ resources:
143
+ - nodes/proxy
144
+ verbs:
145
+ - get
146
+ - apiGroups:
147
+ - ''
148
+ resources:
149
+ - nodes/stats
150
+ - configmaps
151
+ - events
152
+ verbs:
153
+ - create
154
+ - get
155
+ - apiGroups:
156
+ - ''
157
+ resources:
158
+ - configmaps
159
+ verbs:
160
+ - get
161
+ - update
162
+ - create
163
+ resourceNames:
164
+ - otel-container-insight-clusterleader
165
+
166
+ ---
167
+ apiVersion: rbac.authorization.k8s.io/v1
168
+ kind: ClusterRoleBinding
169
+ metadata:
170
+ name: otel-agent-binding
171
+ labels:
172
+ app.kubernetes.io/name: otel-agent
173
+ app.kubernetes.io/managed-by: chant
174
+ app.kubernetes.io/component: rbac
175
+ roleRef:
176
+ apiGroup: rbac.authorization.k8s.io
177
+ kind: ClusterRole
178
+ name: otel-agent-role
179
+ subjects:
180
+ - kind: ServiceAccount
181
+ name: otel-agent-sa
182
+ namespace: observability
183
+
184
+ ---
185
+ apiVersion: v1
186
+ kind: ConfigMap
187
+ metadata:
188
+ name: otel-agent-config
189
+ namespace: observability
190
+ labels:
191
+ app.kubernetes.io/name: otel-agent
192
+ app.kubernetes.io/managed-by: chant
193
+ app.kubernetes.io/component: config
194
+ annotations:
195
+ otel.chant.dev/role: agent
196
+ otel.chant.dev/workload: DaemonSet/otel-agent
197
+ otel.chant.dev/gateways: observability/otel-gateway=loadbalancing,observability/otel-gateway=service
198
+ data:
199
+ config.yaml: |
200
+ receivers:
201
+ otlp:
202
+ protocols:
203
+ grpc:
204
+ endpoint: 0.0.0.0:4317
205
+ http:
206
+ endpoint: 0.0.0.0:4318
207
+
208
+ processors:
209
+ memory_limiter:
210
+ check_interval: 1s
211
+ limit_percentage: 80
212
+ spike_limit_percentage: 20
213
+ batch:
214
+ timeout: 1s
215
+
216
+ exporters:
217
+ loadbalancing/gateway:
218
+ routing_key: traceID
219
+ protocol:
220
+ otlp:
221
+ tls:
222
+ insecure: true
223
+ resolver:
224
+ k8s:
225
+ service: otel-gateway-headless.observability
226
+ ports: [4317]
227
+ otlp/gateway:
228
+ endpoint: otel-gateway.observability.svc:4317
229
+ tls:
230
+ insecure: true
231
+
232
+ extensions:
233
+ health_check:
234
+ endpoint: 0.0.0.0:13133
235
+
236
+ service:
237
+ extensions: [health_check]
238
+ pipelines:
239
+ traces:
240
+ receivers: [otlp]
241
+ processors: [memory_limiter, batch]
242
+ exporters: [loadbalancing/gateway]
243
+ metrics:
244
+ receivers: [otlp]
245
+ processors: [memory_limiter, batch]
246
+ exporters: [otlp/gateway]
247
+ logs:
248
+ receivers: [otlp]
249
+ processors: [memory_limiter, batch]
250
+ exporters: [otlp/gateway]
251
+
252
+ ---
253
+ apiVersion: v1
254
+ kind: Service
255
+ metadata:
256
+ name: otel-agent
257
+ namespace: observability
258
+ labels:
259
+ app.kubernetes.io/name: otel-agent
260
+ app.kubernetes.io/managed-by: chant
261
+ app.kubernetes.io/component: agent
262
+ spec:
263
+ selector:
264
+ app.kubernetes.io/name: otel-agent
265
+ internalTrafficPolicy: Local
266
+ ports:
267
+ - name: otlp-grpc
268
+ port: 4317
269
+ targetPort: otlp-grpc
270
+ protocol: TCP
271
+ - name: otlp-http
272
+ port: 4318
273
+ targetPort: otlp-http
274
+ protocol: TCP
275
+
276
+ ---
277
+ apiVersion: rbac.authorization.k8s.io/v1
278
+ kind: Role
279
+ metadata:
280
+ name: otel-agent-endpoints
281
+ namespace: observability
282
+ labels:
283
+ app.kubernetes.io/name: otel-agent
284
+ app.kubernetes.io/managed-by: chant
285
+ app.kubernetes.io/component: rbac
286
+ rules:
287
+ - apiGroups:
288
+ - ''
289
+ resources:
290
+ - endpoints
291
+ verbs:
292
+ - get
293
+ - list
294
+ - watch
295
+ - apiGroups:
296
+ - discovery.k8s.io
297
+ resources:
298
+ - endpointslices
299
+ verbs:
300
+ - get
301
+ - list
302
+ - watch
303
+
304
+ ---
305
+ apiVersion: rbac.authorization.k8s.io/v1
306
+ kind: RoleBinding
307
+ metadata:
308
+ name: otel-agent-endpoints
309
+ namespace: observability
310
+ labels:
311
+ app.kubernetes.io/name: otel-agent
312
+ app.kubernetes.io/managed-by: chant
313
+ app.kubernetes.io/component: rbac
314
+ roleRef:
315
+ apiGroup: rbac.authorization.k8s.io
316
+ kind: Role
317
+ name: otel-agent-endpoints
318
+ subjects:
319
+ - kind: ServiceAccount
320
+ name: otel-agent-sa
321
+ namespace: observability
322
+
323
+ ---
324
+ apiVersion: networking.k8s.io/v1
325
+ kind: NetworkPolicy
326
+ metadata:
327
+ name: agents-default-deny
328
+ namespace: agents
329
+ labels:
330
+ app.kubernetes.io/name: agents
331
+ app.kubernetes.io/managed-by: chant
332
+ app.kubernetes.io/part-of: agent-observability
333
+ app.kubernetes.io/component: network-policy
334
+ spec:
335
+ podSelector: {}
336
+ policyTypes:
337
+ - Ingress
338
+
339
+ ---
340
+ apiVersion: apps/v1
341
+ kind: Deployment
342
+ metadata:
343
+ name: support-agent
344
+ namespace: agents
345
+ labels:
346
+ app.kubernetes.io/name: support-agent
347
+ app.kubernetes.io/component: agent
348
+ spec:
349
+ replicas: 1
350
+ selector:
351
+ matchLabels:
352
+ app.kubernetes.io/name: support-agent
353
+ template:
354
+ metadata:
355
+ labels:
356
+ app.kubernetes.io/name: support-agent
357
+ app.kubernetes.io/component: agent
358
+ spec:
359
+ securityContext:
360
+ runAsNonRoot: true
361
+ runAsUser: 1000
362
+ runAsGroup: 1000
363
+ containers:
364
+ - name: agent
365
+ image: agent-observability-demo:0.1.0
366
+ imagePullPolicy: IfNotPresent
367
+ env:
368
+ - name: OTLP_ENDPOINT
369
+ value: http://otel-agent.observability.svc:4318
370
+ - name: INTERVAL_MS
371
+ value: '2000'
372
+ resources:
373
+ requests:
374
+ cpu: '20m'
375
+ memory: '64Mi'
376
+ limits:
377
+ cpu: '200m'
378
+ memory: '128Mi'
379
+ securityContext:
380
+ allowPrivilegeEscalation: false
381
+ readOnlyRootFilesystem: true
382
+ capabilities:
383
+ drop:
384
+ - ALL
385
+
386
+ ---
387
+ apiVersion: apps/v1
388
+ kind: Deployment
389
+ metadata:
390
+ name: otel-gateway
391
+ namespace: observability
392
+ labels:
393
+ app.kubernetes.io/name: otel-gateway
394
+ app.kubernetes.io/managed-by: chant
395
+ app.kubernetes.io/component: gateway
396
+ annotations:
397
+ otel.chant.dev/role: gateway
398
+ otel.chant.dev/config: otel-gateway-config
399
+ spec:
400
+ replicas: 2
401
+ selector:
402
+ matchLabels:
403
+ app.kubernetes.io/name: otel-gateway
404
+ template:
405
+ metadata:
406
+ labels:
407
+ app.kubernetes.io/name: otel-gateway
408
+ app.kubernetes.io/component: gateway
409
+ spec:
410
+ serviceAccountName: otel-gateway-sa
411
+ containers:
412
+ - name: otel-gateway
413
+ image: otel/opentelemetry-collector-contrib:0.130.0
414
+ args:
415
+ - --config=/etc/otel/config.yaml
416
+ ports:
417
+ - containerPort: 4317
418
+ name: otlp-grpc
419
+ - containerPort: 13133
420
+ name: health
421
+ resources:
422
+ requests:
423
+ cpu: '200m'
424
+ memory: '512Mi'
425
+ limits:
426
+ cpu: '1'
427
+ memory: '1Gi'
428
+ volumeMounts:
429
+ - name: config
430
+ mountPath: /etc/otel
431
+ readOnly: true
432
+ securityContext:
433
+ runAsNonRoot: true
434
+ runAsUser: 10001
435
+ readOnlyRootFilesystem: true
436
+ allowPrivilegeEscalation: false
437
+ capabilities:
438
+ drop:
439
+ - ALL
440
+ imagePullPolicy: IfNotPresent
441
+ livenessProbe:
442
+ httpGet:
443
+ path: /
444
+ port: health
445
+ readinessProbe:
446
+ httpGet:
447
+ path: /
448
+ port: health
449
+ volumes:
450
+ - name: config
451
+ configMap:
452
+ name: otel-gateway-config
453
+
454
+ ---
455
+ apiVersion: v1
456
+ kind: Service
457
+ metadata:
458
+ name: otel-gateway
459
+ namespace: observability
460
+ labels:
461
+ app.kubernetes.io/name: otel-gateway
462
+ app.kubernetes.io/managed-by: chant
463
+ app.kubernetes.io/component: gateway
464
+ spec:
465
+ type: ClusterIP
466
+ selector:
467
+ app.kubernetes.io/name: otel-gateway
468
+ ports:
469
+ - name: otlp-grpc
470
+ port: 4317
471
+ targetPort: otlp-grpc
472
+ protocol: TCP
473
+
474
+ ---
475
+ apiVersion: v1
476
+ kind: Service
477
+ metadata:
478
+ name: otel-gateway-headless
479
+ namespace: observability
480
+ labels:
481
+ app.kubernetes.io/name: otel-gateway
482
+ app.kubernetes.io/managed-by: chant
483
+ app.kubernetes.io/component: gateway
484
+ spec:
485
+ clusterIP: None
486
+ selector:
487
+ app.kubernetes.io/name: otel-gateway
488
+ ports:
489
+ - name: otlp-grpc
490
+ port: 4317
491
+ targetPort: otlp-grpc
492
+ protocol: TCP
493
+
494
+ ---
495
+ apiVersion: v1
496
+ kind: ServiceAccount
497
+ metadata:
498
+ name: otel-gateway-sa
499
+ namespace: observability
500
+ labels:
501
+ app.kubernetes.io/name: otel-gateway
502
+ app.kubernetes.io/managed-by: chant
503
+ app.kubernetes.io/component: gateway
504
+
505
+ ---
506
+ apiVersion: v1
507
+ kind: ConfigMap
508
+ metadata:
509
+ name: otel-gateway-config
510
+ namespace: observability
511
+ labels:
512
+ app.kubernetes.io/name: otel-gateway
513
+ app.kubernetes.io/managed-by: chant
514
+ app.kubernetes.io/component: config
515
+ annotations:
516
+ otel.chant.dev/role: gateway
517
+ otel.chant.dev/workload: Deployment/otel-gateway
518
+ data:
519
+ config.yaml: |
520
+ # chant: semconv gen_ai github.com/open-telemetry/semantic-conventions@v1.41.1 (transform/genai_content, redaction/genai_content, filter/genai_spans, spanmetrics/genai, sum/genai_tokens)
521
+
522
+ receivers:
523
+ otlp:
524
+ protocols:
525
+ grpc:
526
+ endpoint: 0.0.0.0:4317
527
+
528
+ processors:
529
+ memory_limiter:
530
+ check_interval: 1s
531
+ limit_percentage: 80
532
+ spike_limit_percentage: 20
533
+ transform/genai_content:
534
+ error_mode: ignore
535
+ trace_statements:
536
+ - context: span
537
+ statements: ["delete_key(span.attributes, \"gen_ai.system_instructions\")", "delete_key(span.attributes, \"gen_ai.input.messages\")", "delete_key(span.attributes, \"gen_ai.output.messages\")", "delete_key(span.attributes, \"gen_ai.tool.call.arguments\")", "delete_key(span.attributes, \"gen_ai.tool.call.result\")", "delete_key(span.attributes, \"gen_ai.retrieval.query.text\")", "delete_key(span.attributes, \"gen_ai.retrieval.documents\")", "delete_key(span.attributes, \"gen_ai.prompt\")", "delete_key(span.attributes, \"gen_ai.completion\")", "delete_matching_keys(span.attributes, \"^gen_ai\\\\.(prompt|completion)\\\\.[0-9]+\\\\..+$\")"]
538
+ - context: spanevent
539
+ statements: ["delete_key(spanevent.attributes, \"gen_ai.system_instructions\")", "delete_key(spanevent.attributes, \"gen_ai.input.messages\")", "delete_key(spanevent.attributes, \"gen_ai.output.messages\")", "delete_key(spanevent.attributes, \"gen_ai.tool.call.arguments\")", "delete_key(spanevent.attributes, \"gen_ai.tool.call.result\")", "delete_key(spanevent.attributes, \"gen_ai.retrieval.query.text\")", "delete_key(spanevent.attributes, \"gen_ai.retrieval.documents\")", "delete_key(spanevent.attributes, \"gen_ai.prompt\")", "delete_key(spanevent.attributes, \"gen_ai.completion\")", "delete_matching_keys(spanevent.attributes, \"^gen_ai\\\\.(prompt|completion)\\\\.[0-9]+\\\\..+$\")"]
540
+ log_statements:
541
+ - context: log
542
+ statements: ["delete_key(log.attributes, \"gen_ai.system_instructions\")", "delete_key(log.attributes, \"gen_ai.input.messages\")", "delete_key(log.attributes, \"gen_ai.output.messages\")", "delete_key(log.attributes, \"gen_ai.tool.call.arguments\")", "delete_key(log.attributes, \"gen_ai.tool.call.result\")", "delete_key(log.attributes, \"gen_ai.retrieval.query.text\")", "delete_key(log.attributes, \"gen_ai.retrieval.documents\")", "delete_key(log.attributes, \"gen_ai.prompt\")", "delete_key(log.attributes, \"gen_ai.completion\")", "delete_matching_keys(log.attributes, \"^gen_ai\\\\.(prompt|completion)\\\\.[0-9]+\\\\..+$\")", "delete_matching_keys(log.body, \"^(content|message|tool_calls)$\") where IsMap(log.body) and (IsMatch(log.event_name, \"^(gen_ai\\\\.system\\\\.message|gen_ai\\\\.user\\\\.message|gen_ai\\\\.assistant\\\\.message|gen_ai\\\\.tool\\\\.message|gen_ai\\\\.choice)$\") or IsMatch(log.attributes[\"event.name\"], \"^(gen_ai\\\\.system\\\\.message|gen_ai\\\\.user\\\\.message|gen_ai\\\\.assistant\\\\.message|gen_ai\\\\.tool\\\\.message|gen_ai\\\\.choice)$\"))", "set(log.body, \"\") where IsString(log.body) and (IsMatch(log.event_name, \"^(gen_ai\\\\.system\\\\.message|gen_ai\\\\.user\\\\.message|gen_ai\\\\.assistant\\\\.message|gen_ai\\\\.tool\\\\.message|gen_ai\\\\.choice)$\") or IsMatch(log.attributes[\"event.name\"], \"^(gen_ai\\\\.system\\\\.message|gen_ai\\\\.user\\\\.message|gen_ai\\\\.assistant\\\\.message|gen_ai\\\\.tool\\\\.message|gen_ai\\\\.choice)$\"))"]
543
+ redaction/genai_content:
544
+ allow_all_keys: true
545
+ blocked_key_patterns: [^(gen_ai\.system_instructions|gen_ai\.input\.messages|gen_ai\.output\.messages|gen_ai\.tool\.call\.arguments|gen_ai\.tool\.call\.result|gen_ai\.retrieval\.query\.text|gen_ai\.retrieval\.documents|gen_ai\.prompt|gen_ai\.completion)$, "^gen_ai\\.(prompt|completion)\\.[0-9]+\\..+$"]
546
+ tail_sampling:
547
+ decision_wait: 5s
548
+ num_traces: 50000
549
+ policies:
550
+ - name: errors
551
+ type: status_code
552
+ status_code:
553
+ status_codes: [ERROR]
554
+ - name: slow
555
+ type: latency
556
+ latency:
557
+ threshold_ms: 2000
558
+ - name: baseline
559
+ type: probabilistic
560
+ probabilistic:
561
+ sampling_percentage: 10
562
+ batch:
563
+ timeout: 5s
564
+ filter/genai_spans:
565
+ error_mode: ignore
566
+ traces:
567
+ span: ["attributes[\"gen_ai.operation.name\"] == nil"]
568
+
569
+ exporters:
570
+ otlp/tempo:
571
+ endpoint: tempo:4317
572
+ tls:
573
+ insecure: true
574
+ prometheus:
575
+ endpoint: 0.0.0.0:8889
576
+ metric_expiration: 10m
577
+ otlphttp/loki:
578
+ endpoint: http://loki:3100/otlp
579
+
580
+ connectors:
581
+ spanmetrics:
582
+ histogram:
583
+ unit: ms
584
+ explicit:
585
+ buckets: [50ms, 100ms, 250ms, 500ms, 1s, 2s, 5s, 10s]
586
+ metrics_flush_interval: 15s
587
+ forward/genai: {}
588
+ forward/sampled: {}
589
+ spanmetrics/genai:
590
+ namespace: genai
591
+ dimensions:
592
+ - name: gen_ai.operation.name
593
+ - name: gen_ai.request.model
594
+ - name: gen_ai.tool.name
595
+ - name: error.type
596
+ histogram:
597
+ unit: s
598
+ explicit:
599
+ buckets: [100ms, 250ms, 500ms, 1s, 2s, 5s, 10s, 20s, 40s, 80s]
600
+ metrics_flush_interval: 15s
601
+ sum/genai_tokens:
602
+ spans:
603
+ genai.tokens.input:
604
+ source_attribute: gen_ai.usage.input_tokens
605
+ description: Input tokens used by GenAI operations
606
+ conditions: ["attributes[\"gen_ai.usage.input_tokens\"] != nil and not (kind == SPAN_KIND_INTERNAL and (attributes[\"gen_ai.operation.name\"] == \"invoke_agent\" or attributes[\"gen_ai.operation.name\"] == \"invoke_workflow\"))"]
607
+ attributes:
608
+ - key: gen_ai.request.model
609
+ default_value: unknown
610
+ genai.tokens.output:
611
+ source_attribute: gen_ai.usage.output_tokens
612
+ description: Output tokens used by GenAI operations
613
+ conditions: ["attributes[\"gen_ai.usage.output_tokens\"] != nil and not (kind == SPAN_KIND_INTERNAL and (attributes[\"gen_ai.operation.name\"] == \"invoke_agent\" or attributes[\"gen_ai.operation.name\"] == \"invoke_workflow\"))"]
614
+ attributes:
615
+ - key: gen_ai.request.model
616
+ default_value: unknown
617
+
618
+ extensions:
619
+ health_check:
620
+ endpoint: 0.0.0.0:13133
621
+
622
+ service:
623
+ extensions: [health_check]
624
+ pipelines:
625
+ traces:
626
+ receivers: [otlp]
627
+ processors: [memory_limiter, transform/genai_content, redaction/genai_content]
628
+ exporters: [spanmetrics, forward/genai, forward/sampled]
629
+ traces/sampled:
630
+ receivers: [forward/sampled]
631
+ processors: [tail_sampling, batch]
632
+ exporters: [otlp/tempo]
633
+ traces/genai:
634
+ receivers: [forward/genai]
635
+ processors: [filter/genai_spans]
636
+ exporters: [spanmetrics/genai, sum/genai_tokens]
637
+ metrics:
638
+ receivers: [otlp, spanmetrics, spanmetrics/genai, sum/genai_tokens]
639
+ processors: [memory_limiter, batch]
640
+ exporters: [prometheus]
641
+ logs:
642
+ receivers: [otlp]
643
+ processors: [memory_limiter, transform/genai_content, redaction/genai_content, batch]
644
+ exporters: [otlphttp/loki]
645
+
646
+ ---
647
+ apiVersion: policy/v1
648
+ kind: PodDisruptionBudget
649
+ metadata:
650
+ name: otel-gateway
651
+ namespace: observability
652
+ labels:
653
+ app.kubernetes.io/name: otel-gateway
654
+ app.kubernetes.io/managed-by: chant
655
+ app.kubernetes.io/component: gateway
656
+ spec:
657
+ maxUnavailable: 1
658
+ selector:
659
+ matchLabels:
660
+ app.kubernetes.io/name: otel-gateway
661
+
662
+ ---
663
+ apiVersion: apps/v1
664
+ kind: Deployment
665
+ metadata:
666
+ name: grafana
667
+ namespace: observability
668
+ labels:
669
+ app.kubernetes.io/name: grafana
670
+ app.kubernetes.io/component: dashboards
671
+ spec:
672
+ replicas: 1
673
+ selector:
674
+ matchLabels:
675
+ app.kubernetes.io/name: grafana
676
+ template:
677
+ metadata:
678
+ labels:
679
+ app.kubernetes.io/name: grafana
680
+ app.kubernetes.io/component: dashboards
681
+ spec:
682
+ securityContext:
683
+ fsGroup: 472
684
+ containers:
685
+ - name: grafana
686
+ image: grafana/grafana:12.4.11
687
+ imagePullPolicy: IfNotPresent
688
+ env:
689
+ - name: GF_AUTH_ANONYMOUS_ENABLED
690
+ value: 'true'
691
+ - name: GF_AUTH_ANONYMOUS_ORG_ROLE
692
+ value: Viewer
693
+ - name: GF_ANALYTICS_REPORTING_ENABLED
694
+ value: 'false'
695
+ - name: GF_ANALYTICS_CHECK_FOR_UPDATES
696
+ value: 'false'
697
+ - name: GF_PLUGINS_PREINSTALL_DISABLED
698
+ value: 'true'
699
+ ports:
700
+ - name: http
701
+ containerPort: 3000
702
+ readinessProbe:
703
+ httpGet:
704
+ path: /api/health
705
+ port: http
706
+ initialDelaySeconds: 5
707
+ periodSeconds: 5
708
+ livenessProbe:
709
+ tcpSocket:
710
+ port: http
711
+ initialDelaySeconds: 30
712
+ periodSeconds: 10
713
+ resources:
714
+ requests:
715
+ cpu: '100m'
716
+ memory: '128Mi'
717
+ limits:
718
+ cpu: '1'
719
+ memory: '512Mi'
720
+ securityContext:
721
+ runAsNonRoot: true
722
+ runAsUser: 472
723
+ readOnlyRootFilesystem: true
724
+ allowPrivilegeEscalation: false
725
+ capabilities:
726
+ drop:
727
+ - ALL
728
+ volumeMounts:
729
+ - name: data
730
+ mountPath: /var/lib/grafana
731
+ - name: tmp
732
+ mountPath: /tmp
733
+ - name: grafana-provisioning
734
+ mountPath: /etc/grafana/provisioning
735
+ readOnly: true
736
+ - name: grafana-dashboards-0
737
+ mountPath: /var/lib/grafana/dashboards/Agent observability
738
+ readOnly: true
739
+ volumes:
740
+ - name: data
741
+ emptyDir: {}
742
+ - name: tmp
743
+ emptyDir: {}
744
+ - name: grafana-provisioning
745
+ projected:
746
+ sources:
747
+ - configMap:
748
+ name: grafana-datasources
749
+ items:
750
+ - key: chant.yaml
751
+ path: datasources/chant.yaml
752
+ - configMap:
753
+ name: grafana-dashboard-providers
754
+ items:
755
+ - key: chant.yaml
756
+ path: dashboards/chant.yaml
757
+ - name: grafana-dashboards-0
758
+ projected:
759
+ sources:
760
+ - configMap:
761
+ name: grafana-dashboard-red-traces-span-metrics
762
+ items:
763
+ - key: red-traces-span-metrics.json
764
+ path: red-traces-span-metrics.json
765
+ - configMap:
766
+ name: grafana-dashboard-slo-support-agent-runs
767
+ items:
768
+ - key: slo-support-agent-runs.json
769
+ path: slo-support-agent-runs.json
770
+ - configMap:
771
+ name: grafana-dashboard-genai-agents
772
+ items:
773
+ - key: genai-agents.json
774
+ path: genai-agents.json
775
+
776
+ ---
777
+ apiVersion: v1
778
+ kind: ConfigMap
779
+ metadata:
780
+ name: grafana-dashboard-red-traces-span-metrics
781
+ namespace: observability
782
+ labels:
783
+ app.kubernetes.io/name: grafana
784
+ app.kubernetes.io/component: dashboards
785
+ grafana_dashboard: '1'
786
+ annotations:
787
+ k8s-sidecar-target-directory: Agent observability
788
+ data:
789
+ red-traces-span-metrics.json: |
790
+ {
791
+ "annotations": {
792
+ "list": []
793
+ },
794
+ "description": "Rate, errors and duration per service from the traces.span.metrics span metrics.",
795
+ "editable": true,
796
+ "fiscalYearStartMonth": 0,
797
+ "graphTooltip": 1,
798
+ "links": [],
799
+ "panels": [
800
+ {
801
+ "type": "row",
802
+ "collapsed": false,
803
+ "title": "Rate and errors",
804
+ "gridPos": {
805
+ "h": 1,
806
+ "w": 24,
807
+ "x": 0,
808
+ "y": 0
809
+ },
810
+ "id": 1,
811
+ "panels": []
812
+ },
813
+ {
814
+ "type": "timeseries",
815
+ "id": 2,
816
+ "title": "Rate",
817
+ "description": "Spans per second by service, from traces_span_metrics_calls_total.",
818
+ "gridPos": {
819
+ "h": 8,
820
+ "w": 12,
821
+ "x": 0,
822
+ "y": 1
823
+ },
824
+ "datasource": {
825
+ "type": "prometheus",
826
+ "uid": "prometheus"
827
+ },
828
+ "targets": [
829
+ {
830
+ "datasource": {
831
+ "type": "prometheus",
832
+ "uid": "prometheus"
833
+ },
834
+ "refId": "A",
835
+ "expr": "sum by (service_name) (rate(traces_span_metrics_calls_total{service_name=~\"$service\"}[$__rate_interval]))",
836
+ "legendFormat": "{{service_name}}"
837
+ }
838
+ ],
839
+ "options": {},
840
+ "fieldConfig": {
841
+ "defaults": {
842
+ "unit": "reqps"
843
+ },
844
+ "overrides": []
845
+ }
846
+ },
847
+ {
848
+ "type": "timeseries",
849
+ "id": 3,
850
+ "title": "Errors",
851
+ "description": "Share of spans with status_code=\"STATUS_CODE_ERROR\", by service.",
852
+ "gridPos": {
853
+ "h": 8,
854
+ "w": 12,
855
+ "x": 12,
856
+ "y": 1
857
+ },
858
+ "datasource": {
859
+ "type": "prometheus",
860
+ "uid": "prometheus"
861
+ },
862
+ "targets": [
863
+ {
864
+ "datasource": {
865
+ "type": "prometheus",
866
+ "uid": "prometheus"
867
+ },
868
+ "refId": "A",
869
+ "expr": "(\nsum by (service_name) (rate(traces_span_metrics_calls_total{service_name=~\"$service\", status_code=\"STATUS_CODE_ERROR\"}[$__rate_interval]))\nor\nsum by (service_name) (rate(traces_span_metrics_calls_total{service_name=~\"$service\"}[$__rate_interval])) * 0\n)\n/\nsum by (service_name) (rate(traces_span_metrics_calls_total{service_name=~\"$service\"}[$__rate_interval]))",
870
+ "legendFormat": "{{service_name}}"
871
+ }
872
+ ],
873
+ "options": {},
874
+ "fieldConfig": {
875
+ "defaults": {
876
+ "unit": "percentunit",
877
+ "min": 0
878
+ },
879
+ "overrides": []
880
+ }
881
+ },
882
+ {
883
+ "type": "row",
884
+ "collapsed": false,
885
+ "title": "Duration",
886
+ "gridPos": {
887
+ "h": 1,
888
+ "w": 24,
889
+ "x": 0,
890
+ "y": 9
891
+ },
892
+ "id": 4,
893
+ "panels": []
894
+ },
895
+ {
896
+ "type": "timeseries",
897
+ "id": 5,
898
+ "title": "Duration p50",
899
+ "description": "p50 duration of spans by service, from traces_span_metrics_duration_milliseconds.",
900
+ "gridPos": {
901
+ "h": 8,
902
+ "w": 8,
903
+ "x": 0,
904
+ "y": 10
905
+ },
906
+ "datasource": {
907
+ "type": "prometheus",
908
+ "uid": "prometheus"
909
+ },
910
+ "targets": [
911
+ {
912
+ "datasource": {
913
+ "type": "prometheus",
914
+ "uid": "prometheus"
915
+ },
916
+ "refId": "A",
917
+ "expr": "histogram_quantile(0.5, sum by (le, service_name) (rate(traces_span_metrics_duration_milliseconds_bucket{service_name=~\"$service\"}[$__rate_interval])))",
918
+ "legendFormat": "{{service_name}}"
919
+ }
920
+ ],
921
+ "options": {},
922
+ "fieldConfig": {
923
+ "defaults": {
924
+ "unit": "ms"
925
+ },
926
+ "overrides": []
927
+ }
928
+ },
929
+ {
930
+ "type": "timeseries",
931
+ "id": 6,
932
+ "title": "Duration p95",
933
+ "description": "p95 duration of spans by service, from traces_span_metrics_duration_milliseconds.",
934
+ "gridPos": {
935
+ "h": 8,
936
+ "w": 8,
937
+ "x": 8,
938
+ "y": 10
939
+ },
940
+ "datasource": {
941
+ "type": "prometheus",
942
+ "uid": "prometheus"
943
+ },
944
+ "targets": [
945
+ {
946
+ "datasource": {
947
+ "type": "prometheus",
948
+ "uid": "prometheus"
949
+ },
950
+ "refId": "A",
951
+ "expr": "histogram_quantile(0.95, sum by (le, service_name) (rate(traces_span_metrics_duration_milliseconds_bucket{service_name=~\"$service\"}[$__rate_interval])))",
952
+ "legendFormat": "{{service_name}}"
953
+ }
954
+ ],
955
+ "options": {},
956
+ "fieldConfig": {
957
+ "defaults": {
958
+ "unit": "ms"
959
+ },
960
+ "overrides": []
961
+ }
962
+ },
963
+ {
964
+ "type": "timeseries",
965
+ "id": 7,
966
+ "title": "Duration p99",
967
+ "description": "p99 duration of spans by service, from traces_span_metrics_duration_milliseconds.",
968
+ "gridPos": {
969
+ "h": 8,
970
+ "w": 8,
971
+ "x": 16,
972
+ "y": 10
973
+ },
974
+ "datasource": {
975
+ "type": "prometheus",
976
+ "uid": "prometheus"
977
+ },
978
+ "targets": [
979
+ {
980
+ "datasource": {
981
+ "type": "prometheus",
982
+ "uid": "prometheus"
983
+ },
984
+ "refId": "A",
985
+ "expr": "histogram_quantile(0.99, sum by (le, service_name) (rate(traces_span_metrics_duration_milliseconds_bucket{service_name=~\"$service\"}[$__rate_interval])))",
986
+ "legendFormat": "{{service_name}}"
987
+ }
988
+ ],
989
+ "options": {},
990
+ "fieldConfig": {
991
+ "defaults": {
992
+ "unit": "ms"
993
+ },
994
+ "overrides": []
995
+ }
996
+ }
997
+ ],
998
+ "refresh": "1m",
999
+ "schemaVersion": 42,
1000
+ "tags": [
1001
+ "red",
1002
+ "spanmetrics"
1003
+ ],
1004
+ "templating": {
1005
+ "list": [
1006
+ {
1007
+ "type": "query",
1008
+ "name": "service",
1009
+ "label": "Service",
1010
+ "datasource": {
1011
+ "type": "prometheus",
1012
+ "uid": "prometheus"
1013
+ },
1014
+ "query": "label_values(traces_span_metrics_calls_total, service_name)",
1015
+ "definition": "label_values(traces_span_metrics_calls_total, service_name)",
1016
+ "refresh": 2,
1017
+ "sort": 1,
1018
+ "multi": true,
1019
+ "includeAll": true,
1020
+ "allValue": ".*",
1021
+ "options": []
1022
+ }
1023
+ ]
1024
+ },
1025
+ "time": {
1026
+ "from": "now-6h",
1027
+ "to": "now"
1028
+ },
1029
+ "timepicker": {},
1030
+ "timezone": "browser",
1031
+ "title": "Services: rate, errors, duration",
1032
+ "uid": "red-traces-span-metrics",
1033
+ "weekStart": ""
1034
+ }
1035
+
1036
+ ---
1037
+ apiVersion: v1
1038
+ kind: ConfigMap
1039
+ metadata:
1040
+ name: grafana-dashboard-slo-support-agent-runs
1041
+ namespace: observability
1042
+ labels:
1043
+ app.kubernetes.io/name: grafana
1044
+ app.kubernetes.io/component: dashboards
1045
+ grafana_dashboard: '1'
1046
+ annotations:
1047
+ k8s-sidecar-target-directory: Agent observability
1048
+ data:
1049
+ slo-support-agent-runs.json: |
1050
+ {
1051
+ "annotations": {
1052
+ "list": []
1053
+ },
1054
+ "description": "99% over 28d: SLI, error budget and burn rates for the support-agent-runs SLO.",
1055
+ "editable": true,
1056
+ "fiscalYearStartMonth": 0,
1057
+ "graphTooltip": 1,
1058
+ "links": [],
1059
+ "panels": [
1060
+ {
1061
+ "type": "row",
1062
+ "collapsed": false,
1063
+ "title": "Objective",
1064
+ "gridPos": {
1065
+ "h": 1,
1066
+ "w": 24,
1067
+ "x": 0,
1068
+ "y": 0
1069
+ },
1070
+ "id": 1,
1071
+ "panels": []
1072
+ },
1073
+ {
1074
+ "type": "stat",
1075
+ "id": 2,
1076
+ "title": "SLI over 28d",
1077
+ "description": "Share of good events over the last 28d. Green at or above the 99% objective.",
1078
+ "gridPos": {
1079
+ "h": 5,
1080
+ "w": 6,
1081
+ "x": 0,
1082
+ "y": 1
1083
+ },
1084
+ "datasource": {
1085
+ "type": "prometheus",
1086
+ "uid": "prometheus"
1087
+ },
1088
+ "targets": [
1089
+ {
1090
+ "datasource": {
1091
+ "type": "prometheus",
1092
+ "uid": "prometheus"
1093
+ },
1094
+ "refId": "A",
1095
+ "expr": "1 - slo:sli_error:ratio_rate28d{slo=\"support-agent-runs\"}",
1096
+ "legendFormat": "SLI"
1097
+ }
1098
+ ],
1099
+ "options": {
1100
+ "colorMode": "background",
1101
+ "graphMode": "none",
1102
+ "reduceOptions": {
1103
+ "calcs": [
1104
+ "lastNotNull"
1105
+ ]
1106
+ }
1107
+ },
1108
+ "fieldConfig": {
1109
+ "defaults": {
1110
+ "unit": "percentunit",
1111
+ "decimals": 3,
1112
+ "thresholds": {
1113
+ "mode": "absolute",
1114
+ "steps": [
1115
+ {
1116
+ "value": null,
1117
+ "color": "red"
1118
+ },
1119
+ {
1120
+ "value": 0.99,
1121
+ "color": "green"
1122
+ }
1123
+ ]
1124
+ }
1125
+ },
1126
+ "overrides": []
1127
+ }
1128
+ },
1129
+ {
1130
+ "type": "stat",
1131
+ "id": 3,
1132
+ "title": "Objective",
1133
+ "gridPos": {
1134
+ "h": 5,
1135
+ "w": 6,
1136
+ "x": 6,
1137
+ "y": 1
1138
+ },
1139
+ "datasource": {
1140
+ "type": "prometheus",
1141
+ "uid": "prometheus"
1142
+ },
1143
+ "targets": [
1144
+ {
1145
+ "datasource": {
1146
+ "type": "prometheus",
1147
+ "uid": "prometheus"
1148
+ },
1149
+ "refId": "A",
1150
+ "expr": "slo:objective:ratio{slo=\"support-agent-runs\"}",
1151
+ "legendFormat": "objective"
1152
+ }
1153
+ ],
1154
+ "options": {
1155
+ "colorMode": "none",
1156
+ "graphMode": "none",
1157
+ "reduceOptions": {
1158
+ "calcs": [
1159
+ "lastNotNull"
1160
+ ]
1161
+ }
1162
+ },
1163
+ "fieldConfig": {
1164
+ "defaults": {
1165
+ "unit": "percentunit",
1166
+ "decimals": 3
1167
+ },
1168
+ "overrides": []
1169
+ }
1170
+ },
1171
+ {
1172
+ "type": "stat",
1173
+ "id": 4,
1174
+ "title": "Error budget remaining",
1175
+ "description": "Share of the 28d error budget (1% of events) not yet spent. Below 0 the objective is missed.",
1176
+ "gridPos": {
1177
+ "h": 5,
1178
+ "w": 6,
1179
+ "x": 12,
1180
+ "y": 1
1181
+ },
1182
+ "datasource": {
1183
+ "type": "prometheus",
1184
+ "uid": "prometheus"
1185
+ },
1186
+ "targets": [
1187
+ {
1188
+ "datasource": {
1189
+ "type": "prometheus",
1190
+ "uid": "prometheus"
1191
+ },
1192
+ "refId": "A",
1193
+ "expr": "slo:error_budget:remaining{slo=\"support-agent-runs\"}",
1194
+ "legendFormat": "remaining"
1195
+ }
1196
+ ],
1197
+ "options": {
1198
+ "colorMode": "background",
1199
+ "graphMode": "area",
1200
+ "reduceOptions": {
1201
+ "calcs": [
1202
+ "lastNotNull"
1203
+ ]
1204
+ }
1205
+ },
1206
+ "fieldConfig": {
1207
+ "defaults": {
1208
+ "unit": "percentunit",
1209
+ "decimals": 1,
1210
+ "thresholds": {
1211
+ "mode": "absolute",
1212
+ "steps": [
1213
+ {
1214
+ "value": null,
1215
+ "color": "red"
1216
+ },
1217
+ {
1218
+ "value": 0,
1219
+ "color": "orange"
1220
+ },
1221
+ {
1222
+ "value": 0.25,
1223
+ "color": "green"
1224
+ }
1225
+ ]
1226
+ }
1227
+ },
1228
+ "overrides": []
1229
+ }
1230
+ },
1231
+ {
1232
+ "type": "stat",
1233
+ "id": 5,
1234
+ "title": "Burn-rate alerts firing",
1235
+ "description": "ErrorBudgetBurn alerts for this SLO, any tier.",
1236
+ "gridPos": {
1237
+ "h": 5,
1238
+ "w": 6,
1239
+ "x": 18,
1240
+ "y": 1
1241
+ },
1242
+ "datasource": {
1243
+ "type": "prometheus",
1244
+ "uid": "prometheus"
1245
+ },
1246
+ "targets": [
1247
+ {
1248
+ "datasource": {
1249
+ "type": "prometheus",
1250
+ "uid": "prometheus"
1251
+ },
1252
+ "refId": "A",
1253
+ "expr": "sum(ALERTS{alertname=\"ErrorBudgetBurn\", slo=\"support-agent-runs\", alertstate=\"firing\"}) or vector(0)",
1254
+ "legendFormat": "firing"
1255
+ }
1256
+ ],
1257
+ "options": {
1258
+ "colorMode": "background",
1259
+ "graphMode": "none",
1260
+ "reduceOptions": {
1261
+ "calcs": [
1262
+ "lastNotNull"
1263
+ ]
1264
+ }
1265
+ },
1266
+ "fieldConfig": {
1267
+ "defaults": {
1268
+ "unit": "none",
1269
+ "decimals": 0,
1270
+ "thresholds": {
1271
+ "mode": "absolute",
1272
+ "steps": [
1273
+ {
1274
+ "value": null,
1275
+ "color": "green"
1276
+ },
1277
+ {
1278
+ "value": 1,
1279
+ "color": "red"
1280
+ }
1281
+ ]
1282
+ }
1283
+ },
1284
+ "overrides": []
1285
+ }
1286
+ },
1287
+ {
1288
+ "type": "timeseries",
1289
+ "id": 6,
1290
+ "title": "SLI over 28d",
1291
+ "description": "The dashed line is the 99% objective.",
1292
+ "gridPos": {
1293
+ "h": 8,
1294
+ "w": 12,
1295
+ "x": 0,
1296
+ "y": 6
1297
+ },
1298
+ "datasource": {
1299
+ "type": "prometheus",
1300
+ "uid": "prometheus"
1301
+ },
1302
+ "targets": [
1303
+ {
1304
+ "datasource": {
1305
+ "type": "prometheus",
1306
+ "uid": "prometheus"
1307
+ },
1308
+ "refId": "A",
1309
+ "expr": "1 - slo:sli_error:ratio_rate28d{slo=\"support-agent-runs\"}",
1310
+ "legendFormat": "SLI"
1311
+ }
1312
+ ],
1313
+ "options": {},
1314
+ "fieldConfig": {
1315
+ "defaults": {
1316
+ "unit": "percentunit",
1317
+ "decimals": 3,
1318
+ "thresholds": {
1319
+ "mode": "absolute",
1320
+ "steps": [
1321
+ {
1322
+ "value": null,
1323
+ "color": "red"
1324
+ },
1325
+ {
1326
+ "value": 0.99,
1327
+ "color": "green"
1328
+ }
1329
+ ]
1330
+ },
1331
+ "custom": {
1332
+ "thresholdsStyle": {
1333
+ "mode": "dashed"
1334
+ }
1335
+ }
1336
+ },
1337
+ "overrides": []
1338
+ }
1339
+ },
1340
+ {
1341
+ "type": "timeseries",
1342
+ "id": 7,
1343
+ "title": "Error budget remaining",
1344
+ "gridPos": {
1345
+ "h": 8,
1346
+ "w": 12,
1347
+ "x": 12,
1348
+ "y": 6
1349
+ },
1350
+ "datasource": {
1351
+ "type": "prometheus",
1352
+ "uid": "prometheus"
1353
+ },
1354
+ "targets": [
1355
+ {
1356
+ "datasource": {
1357
+ "type": "prometheus",
1358
+ "uid": "prometheus"
1359
+ },
1360
+ "refId": "A",
1361
+ "expr": "slo:error_budget:remaining{slo=\"support-agent-runs\"}",
1362
+ "legendFormat": "remaining"
1363
+ }
1364
+ ],
1365
+ "options": {},
1366
+ "fieldConfig": {
1367
+ "defaults": {
1368
+ "unit": "percentunit",
1369
+ "decimals": 1,
1370
+ "thresholds": {
1371
+ "mode": "absolute",
1372
+ "steps": [
1373
+ {
1374
+ "value": null,
1375
+ "color": "red"
1376
+ },
1377
+ {
1378
+ "value": 0,
1379
+ "color": "green"
1380
+ }
1381
+ ]
1382
+ },
1383
+ "custom": {
1384
+ "thresholdsStyle": {
1385
+ "mode": "dashed"
1386
+ }
1387
+ }
1388
+ },
1389
+ "overrides": []
1390
+ }
1391
+ },
1392
+ {
1393
+ "type": "row",
1394
+ "collapsed": false,
1395
+ "title": "Burn rate by alert window",
1396
+ "gridPos": {
1397
+ "h": 1,
1398
+ "w": 24,
1399
+ "x": 0,
1400
+ "y": 14
1401
+ },
1402
+ "id": 8,
1403
+ "panels": []
1404
+ },
1405
+ {
1406
+ "type": "timeseries",
1407
+ "id": 9,
1408
+ "title": "Burn rate 1h / 5m (page)",
1409
+ "description": "Error ratio over the budget: 1 spends the 28d budget exactly over 28d. ErrorBudgetBurn fires with severity page while both windows are above 13.44, which spends the budget in 2d2h.",
1410
+ "gridPos": {
1411
+ "h": 8,
1412
+ "w": 12,
1413
+ "x": 0,
1414
+ "y": 15
1415
+ },
1416
+ "datasource": {
1417
+ "type": "prometheus",
1418
+ "uid": "prometheus"
1419
+ },
1420
+ "targets": [
1421
+ {
1422
+ "datasource": {
1423
+ "type": "prometheus",
1424
+ "uid": "prometheus"
1425
+ },
1426
+ "refId": "A",
1427
+ "expr": "slo:sli_error:ratio_rate1h{slo=\"support-agent-runs\"} / 0.01",
1428
+ "legendFormat": "1h"
1429
+ },
1430
+ {
1431
+ "datasource": {
1432
+ "type": "prometheus",
1433
+ "uid": "prometheus"
1434
+ },
1435
+ "refId": "B",
1436
+ "expr": "slo:sli_error:ratio_rate5m{slo=\"support-agent-runs\"} / 0.01",
1437
+ "legendFormat": "5m"
1438
+ }
1439
+ ],
1440
+ "options": {},
1441
+ "fieldConfig": {
1442
+ "defaults": {
1443
+ "unit": "suffix:x",
1444
+ "decimals": 2,
1445
+ "min": 0,
1446
+ "thresholds": {
1447
+ "mode": "absolute",
1448
+ "steps": [
1449
+ {
1450
+ "value": null,
1451
+ "color": "green"
1452
+ },
1453
+ {
1454
+ "value": 13.44,
1455
+ "color": "red"
1456
+ }
1457
+ ]
1458
+ },
1459
+ "custom": {
1460
+ "thresholdsStyle": {
1461
+ "mode": "dashed"
1462
+ }
1463
+ }
1464
+ },
1465
+ "overrides": []
1466
+ }
1467
+ },
1468
+ {
1469
+ "type": "timeseries",
1470
+ "id": 10,
1471
+ "title": "Burn rate 6h / 30m (page)",
1472
+ "description": "Error ratio over the budget: 1 spends the 28d budget exactly over 28d. ErrorBudgetBurn fires with severity page while both windows are above 5.6, which spends the budget in 5d.",
1473
+ "gridPos": {
1474
+ "h": 8,
1475
+ "w": 12,
1476
+ "x": 12,
1477
+ "y": 15
1478
+ },
1479
+ "datasource": {
1480
+ "type": "prometheus",
1481
+ "uid": "prometheus"
1482
+ },
1483
+ "targets": [
1484
+ {
1485
+ "datasource": {
1486
+ "type": "prometheus",
1487
+ "uid": "prometheus"
1488
+ },
1489
+ "refId": "A",
1490
+ "expr": "slo:sli_error:ratio_rate6h{slo=\"support-agent-runs\"} / 0.01",
1491
+ "legendFormat": "6h"
1492
+ },
1493
+ {
1494
+ "datasource": {
1495
+ "type": "prometheus",
1496
+ "uid": "prometheus"
1497
+ },
1498
+ "refId": "B",
1499
+ "expr": "slo:sli_error:ratio_rate30m{slo=\"support-agent-runs\"} / 0.01",
1500
+ "legendFormat": "30m"
1501
+ }
1502
+ ],
1503
+ "options": {},
1504
+ "fieldConfig": {
1505
+ "defaults": {
1506
+ "unit": "suffix:x",
1507
+ "decimals": 2,
1508
+ "min": 0,
1509
+ "thresholds": {
1510
+ "mode": "absolute",
1511
+ "steps": [
1512
+ {
1513
+ "value": null,
1514
+ "color": "green"
1515
+ },
1516
+ {
1517
+ "value": 5.6,
1518
+ "color": "red"
1519
+ }
1520
+ ]
1521
+ },
1522
+ "custom": {
1523
+ "thresholdsStyle": {
1524
+ "mode": "dashed"
1525
+ }
1526
+ }
1527
+ },
1528
+ "overrides": []
1529
+ }
1530
+ },
1531
+ {
1532
+ "type": "timeseries",
1533
+ "id": 11,
1534
+ "title": "Burn rate 1d / 2h (ticket)",
1535
+ "description": "Error ratio over the budget: 1 spends the 28d budget exactly over 28d. ErrorBudgetBurn fires with severity ticket while both windows are above 2.8, which spends the budget in 1w3d.",
1536
+ "gridPos": {
1537
+ "h": 8,
1538
+ "w": 12,
1539
+ "x": 0,
1540
+ "y": 23
1541
+ },
1542
+ "datasource": {
1543
+ "type": "prometheus",
1544
+ "uid": "prometheus"
1545
+ },
1546
+ "targets": [
1547
+ {
1548
+ "datasource": {
1549
+ "type": "prometheus",
1550
+ "uid": "prometheus"
1551
+ },
1552
+ "refId": "A",
1553
+ "expr": "slo:sli_error:ratio_rate1d{slo=\"support-agent-runs\"} / 0.01",
1554
+ "legendFormat": "1d"
1555
+ },
1556
+ {
1557
+ "datasource": {
1558
+ "type": "prometheus",
1559
+ "uid": "prometheus"
1560
+ },
1561
+ "refId": "B",
1562
+ "expr": "slo:sli_error:ratio_rate2h{slo=\"support-agent-runs\"} / 0.01",
1563
+ "legendFormat": "2h"
1564
+ }
1565
+ ],
1566
+ "options": {},
1567
+ "fieldConfig": {
1568
+ "defaults": {
1569
+ "unit": "suffix:x",
1570
+ "decimals": 2,
1571
+ "min": 0,
1572
+ "thresholds": {
1573
+ "mode": "absolute",
1574
+ "steps": [
1575
+ {
1576
+ "value": null,
1577
+ "color": "green"
1578
+ },
1579
+ {
1580
+ "value": 2.8,
1581
+ "color": "red"
1582
+ }
1583
+ ]
1584
+ },
1585
+ "custom": {
1586
+ "thresholdsStyle": {
1587
+ "mode": "dashed"
1588
+ }
1589
+ }
1590
+ },
1591
+ "overrides": []
1592
+ }
1593
+ },
1594
+ {
1595
+ "type": "timeseries",
1596
+ "id": 12,
1597
+ "title": "Burn rate 3d / 6h (ticket)",
1598
+ "description": "Error ratio over the budget: 1 spends the 28d budget exactly over 28d. ErrorBudgetBurn fires with severity ticket while both windows are above 0.933333, which spends the budget in 4w2d.",
1599
+ "gridPos": {
1600
+ "h": 8,
1601
+ "w": 12,
1602
+ "x": 12,
1603
+ "y": 23
1604
+ },
1605
+ "datasource": {
1606
+ "type": "prometheus",
1607
+ "uid": "prometheus"
1608
+ },
1609
+ "targets": [
1610
+ {
1611
+ "datasource": {
1612
+ "type": "prometheus",
1613
+ "uid": "prometheus"
1614
+ },
1615
+ "refId": "A",
1616
+ "expr": "slo:sli_error:ratio_rate3d{slo=\"support-agent-runs\"} / 0.01",
1617
+ "legendFormat": "3d"
1618
+ },
1619
+ {
1620
+ "datasource": {
1621
+ "type": "prometheus",
1622
+ "uid": "prometheus"
1623
+ },
1624
+ "refId": "B",
1625
+ "expr": "slo:sli_error:ratio_rate6h{slo=\"support-agent-runs\"} / 0.01",
1626
+ "legendFormat": "6h"
1627
+ }
1628
+ ],
1629
+ "options": {},
1630
+ "fieldConfig": {
1631
+ "defaults": {
1632
+ "unit": "suffix:x",
1633
+ "decimals": 2,
1634
+ "min": 0,
1635
+ "thresholds": {
1636
+ "mode": "absolute",
1637
+ "steps": [
1638
+ {
1639
+ "value": null,
1640
+ "color": "green"
1641
+ },
1642
+ {
1643
+ "value": 0.933333,
1644
+ "color": "red"
1645
+ }
1646
+ ]
1647
+ },
1648
+ "custom": {
1649
+ "thresholdsStyle": {
1650
+ "mode": "dashed"
1651
+ }
1652
+ }
1653
+ },
1654
+ "overrides": []
1655
+ }
1656
+ }
1657
+ ],
1658
+ "refresh": "1m",
1659
+ "schemaVersion": 42,
1660
+ "tags": [
1661
+ "slo",
1662
+ "support-agent-runs"
1663
+ ],
1664
+ "templating": {
1665
+ "list": []
1666
+ },
1667
+ "time": {
1668
+ "from": "now-7d",
1669
+ "to": "now"
1670
+ },
1671
+ "timepicker": {},
1672
+ "timezone": "browser",
1673
+ "title": "SLO: support-agent-runs",
1674
+ "uid": "slo-support-agent-runs",
1675
+ "weekStart": ""
1676
+ }
1677
+
1678
+ ---
1679
+ apiVersion: v1
1680
+ kind: ConfigMap
1681
+ metadata:
1682
+ name: grafana-dashboard-genai-agents
1683
+ namespace: observability
1684
+ labels:
1685
+ app.kubernetes.io/name: grafana
1686
+ app.kubernetes.io/component: dashboards
1687
+ grafana_dashboard: '1'
1688
+ annotations:
1689
+ k8s-sidecar-target-directory: Agent observability
1690
+ data:
1691
+ genai-agents.json: |
1692
+ {
1693
+ "annotations": {
1694
+ "list": []
1695
+ },
1696
+ "description": "Calls, errors and latency per model and tool, and token usage per model, from the genai GenAI metrics.",
1697
+ "editable": true,
1698
+ "fiscalYearStartMonth": 0,
1699
+ "graphTooltip": 1,
1700
+ "links": [],
1701
+ "panels": [
1702
+ {
1703
+ "type": "row",
1704
+ "collapsed": false,
1705
+ "title": "Models",
1706
+ "gridPos": {
1707
+ "h": 1,
1708
+ "w": 24,
1709
+ "x": 0,
1710
+ "y": 0
1711
+ },
1712
+ "id": 1,
1713
+ "panels": []
1714
+ },
1715
+ {
1716
+ "type": "timeseries",
1717
+ "id": 2,
1718
+ "title": "Calls by model",
1719
+ "description": "GenAI operations per second, from genai_calls_total.",
1720
+ "gridPos": {
1721
+ "h": 8,
1722
+ "w": 8,
1723
+ "x": 0,
1724
+ "y": 1
1725
+ },
1726
+ "datasource": {
1727
+ "type": "prometheus",
1728
+ "uid": "prometheus"
1729
+ },
1730
+ "targets": [
1731
+ {
1732
+ "datasource": {
1733
+ "type": "prometheus",
1734
+ "uid": "prometheus"
1735
+ },
1736
+ "refId": "A",
1737
+ "expr": "sum by (gen_ai_request_model) (rate(genai_calls_total{service_name=~\"$service\", gen_ai_request_model=~\"$model\"}[$__rate_interval]))",
1738
+ "legendFormat": "{{gen_ai_request_model}}"
1739
+ }
1740
+ ],
1741
+ "options": {},
1742
+ "fieldConfig": {
1743
+ "defaults": {
1744
+ "unit": "reqps"
1745
+ },
1746
+ "overrides": []
1747
+ }
1748
+ },
1749
+ {
1750
+ "type": "timeseries",
1751
+ "id": 3,
1752
+ "title": "Errors by model",
1753
+ "description": "Share of operations that ended with status_code=\"STATUS_CODE_ERROR\".",
1754
+ "gridPos": {
1755
+ "h": 8,
1756
+ "w": 8,
1757
+ "x": 8,
1758
+ "y": 1
1759
+ },
1760
+ "datasource": {
1761
+ "type": "prometheus",
1762
+ "uid": "prometheus"
1763
+ },
1764
+ "targets": [
1765
+ {
1766
+ "datasource": {
1767
+ "type": "prometheus",
1768
+ "uid": "prometheus"
1769
+ },
1770
+ "refId": "A",
1771
+ "expr": "(\nsum by (gen_ai_request_model) (rate(genai_calls_total{service_name=~\"$service\", gen_ai_request_model=~\"$model\", status_code=\"STATUS_CODE_ERROR\"}[$__rate_interval]))\nor\nsum by (gen_ai_request_model) (rate(genai_calls_total{service_name=~\"$service\", gen_ai_request_model=~\"$model\"}[$__rate_interval])) * 0\n)\n/\nsum by (gen_ai_request_model) (rate(genai_calls_total{service_name=~\"$service\", gen_ai_request_model=~\"$model\"}[$__rate_interval]))",
1772
+ "legendFormat": "{{gen_ai_request_model}}"
1773
+ }
1774
+ ],
1775
+ "options": {},
1776
+ "fieldConfig": {
1777
+ "defaults": {
1778
+ "unit": "percentunit",
1779
+ "min": 0
1780
+ },
1781
+ "overrides": []
1782
+ }
1783
+ },
1784
+ {
1785
+ "type": "timeseries",
1786
+ "id": 4,
1787
+ "title": "Latency p95 by model",
1788
+ "description": "From genai_duration_seconds.",
1789
+ "gridPos": {
1790
+ "h": 8,
1791
+ "w": 8,
1792
+ "x": 16,
1793
+ "y": 1
1794
+ },
1795
+ "datasource": {
1796
+ "type": "prometheus",
1797
+ "uid": "prometheus"
1798
+ },
1799
+ "targets": [
1800
+ {
1801
+ "datasource": {
1802
+ "type": "prometheus",
1803
+ "uid": "prometheus"
1804
+ },
1805
+ "refId": "A",
1806
+ "expr": "histogram_quantile(0.95, sum by (le, gen_ai_request_model) (rate(genai_duration_seconds_bucket{service_name=~\"$service\", gen_ai_request_model=~\"$model\"}[$__rate_interval])))",
1807
+ "legendFormat": "{{gen_ai_request_model}}"
1808
+ }
1809
+ ],
1810
+ "options": {},
1811
+ "fieldConfig": {
1812
+ "defaults": {
1813
+ "unit": "s"
1814
+ },
1815
+ "overrides": []
1816
+ }
1817
+ },
1818
+ {
1819
+ "type": "row",
1820
+ "collapsed": false,
1821
+ "title": "Tools",
1822
+ "gridPos": {
1823
+ "h": 1,
1824
+ "w": 24,
1825
+ "x": 0,
1826
+ "y": 9
1827
+ },
1828
+ "id": 5,
1829
+ "panels": []
1830
+ },
1831
+ {
1832
+ "type": "timeseries",
1833
+ "id": 6,
1834
+ "title": "Tool calls",
1835
+ "description": "Operations with a gen_ai.tool.name, per second.",
1836
+ "gridPos": {
1837
+ "h": 8,
1838
+ "w": 8,
1839
+ "x": 0,
1840
+ "y": 10
1841
+ },
1842
+ "datasource": {
1843
+ "type": "prometheus",
1844
+ "uid": "prometheus"
1845
+ },
1846
+ "targets": [
1847
+ {
1848
+ "datasource": {
1849
+ "type": "prometheus",
1850
+ "uid": "prometheus"
1851
+ },
1852
+ "refId": "A",
1853
+ "expr": "sum by (gen_ai_tool_name) (rate(genai_calls_total{service_name=~\"$service\", gen_ai_request_model=~\"$model\", gen_ai_tool_name!=\"\"}[$__rate_interval]))",
1854
+ "legendFormat": "{{gen_ai_tool_name}}"
1855
+ }
1856
+ ],
1857
+ "options": {},
1858
+ "fieldConfig": {
1859
+ "defaults": {
1860
+ "unit": "reqps"
1861
+ },
1862
+ "overrides": []
1863
+ }
1864
+ },
1865
+ {
1866
+ "type": "timeseries",
1867
+ "id": 7,
1868
+ "title": "Errors by tool",
1869
+ "gridPos": {
1870
+ "h": 8,
1871
+ "w": 8,
1872
+ "x": 8,
1873
+ "y": 10
1874
+ },
1875
+ "datasource": {
1876
+ "type": "prometheus",
1877
+ "uid": "prometheus"
1878
+ },
1879
+ "targets": [
1880
+ {
1881
+ "datasource": {
1882
+ "type": "prometheus",
1883
+ "uid": "prometheus"
1884
+ },
1885
+ "refId": "A",
1886
+ "expr": "(\nsum by (gen_ai_tool_name) (rate(genai_calls_total{service_name=~\"$service\", gen_ai_request_model=~\"$model\", gen_ai_tool_name!=\"\", status_code=\"STATUS_CODE_ERROR\"}[$__rate_interval]))\nor\nsum by (gen_ai_tool_name) (rate(genai_calls_total{service_name=~\"$service\", gen_ai_request_model=~\"$model\", gen_ai_tool_name!=\"\"}[$__rate_interval])) * 0\n)\n/\nsum by (gen_ai_tool_name) (rate(genai_calls_total{service_name=~\"$service\", gen_ai_request_model=~\"$model\", gen_ai_tool_name!=\"\"}[$__rate_interval]))",
1887
+ "legendFormat": "{{gen_ai_tool_name}}"
1888
+ }
1889
+ ],
1890
+ "options": {},
1891
+ "fieldConfig": {
1892
+ "defaults": {
1893
+ "unit": "percentunit",
1894
+ "min": 0
1895
+ },
1896
+ "overrides": []
1897
+ }
1898
+ },
1899
+ {
1900
+ "type": "timeseries",
1901
+ "id": 8,
1902
+ "title": "Latency p95 by tool",
1903
+ "gridPos": {
1904
+ "h": 8,
1905
+ "w": 8,
1906
+ "x": 16,
1907
+ "y": 10
1908
+ },
1909
+ "datasource": {
1910
+ "type": "prometheus",
1911
+ "uid": "prometheus"
1912
+ },
1913
+ "targets": [
1914
+ {
1915
+ "datasource": {
1916
+ "type": "prometheus",
1917
+ "uid": "prometheus"
1918
+ },
1919
+ "refId": "A",
1920
+ "expr": "histogram_quantile(0.95, sum by (le, gen_ai_tool_name) (rate(genai_duration_seconds_bucket{service_name=~\"$service\", gen_ai_request_model=~\"$model\", gen_ai_tool_name!=\"\"}[$__rate_interval])))",
1921
+ "legendFormat": "{{gen_ai_tool_name}}"
1922
+ }
1923
+ ],
1924
+ "options": {},
1925
+ "fieldConfig": {
1926
+ "defaults": {
1927
+ "unit": "s"
1928
+ },
1929
+ "overrides": []
1930
+ }
1931
+ },
1932
+ {
1933
+ "type": "row",
1934
+ "collapsed": false,
1935
+ "title": "Errors",
1936
+ "gridPos": {
1937
+ "h": 1,
1938
+ "w": 24,
1939
+ "x": 0,
1940
+ "y": 18
1941
+ },
1942
+ "id": 9,
1943
+ "panels": []
1944
+ },
1945
+ {
1946
+ "type": "timeseries",
1947
+ "id": 10,
1948
+ "title": "Errors by type",
1949
+ "description": "Failed operations per second by error.type.",
1950
+ "gridPos": {
1951
+ "h": 8,
1952
+ "w": 24,
1953
+ "x": 0,
1954
+ "y": 19
1955
+ },
1956
+ "datasource": {
1957
+ "type": "prometheus",
1958
+ "uid": "prometheus"
1959
+ },
1960
+ "targets": [
1961
+ {
1962
+ "datasource": {
1963
+ "type": "prometheus",
1964
+ "uid": "prometheus"
1965
+ },
1966
+ "refId": "A",
1967
+ "expr": "sum by (error_type) (rate(genai_calls_total{service_name=~\"$service\", gen_ai_request_model=~\"$model\", status_code=\"STATUS_CODE_ERROR\"}[$__rate_interval]))",
1968
+ "legendFormat": "{{error_type}}"
1969
+ }
1970
+ ],
1971
+ "options": {},
1972
+ "fieldConfig": {
1973
+ "defaults": {
1974
+ "unit": "reqps"
1975
+ },
1976
+ "overrides": []
1977
+ }
1978
+ },
1979
+ {
1980
+ "type": "row",
1981
+ "collapsed": false,
1982
+ "title": "Tokens",
1983
+ "gridPos": {
1984
+ "h": 1,
1985
+ "w": 24,
1986
+ "x": 0,
1987
+ "y": 27
1988
+ },
1989
+ "id": 11,
1990
+ "panels": []
1991
+ },
1992
+ {
1993
+ "type": "timeseries",
1994
+ "id": 12,
1995
+ "title": "Input tokens by model",
1996
+ "description": "Tokens per second by model, from genai_tokens_input_total.",
1997
+ "gridPos": {
1998
+ "h": 8,
1999
+ "w": 8,
2000
+ "x": 0,
2001
+ "y": 28
2002
+ },
2003
+ "datasource": {
2004
+ "type": "prometheus",
2005
+ "uid": "prometheus"
2006
+ },
2007
+ "targets": [
2008
+ {
2009
+ "datasource": {
2010
+ "type": "prometheus",
2011
+ "uid": "prometheus"
2012
+ },
2013
+ "refId": "A",
2014
+ "expr": "sum by (gen_ai_request_model) (rate(genai_tokens_input_total{gen_ai_request_model=~\"$model\"}[$__rate_interval]))",
2015
+ "legendFormat": "{{gen_ai_request_model}}"
2016
+ }
2017
+ ],
2018
+ "options": {
2019
+ "tooltip": {
2020
+ "mode": "multi"
2021
+ }
2022
+ },
2023
+ "fieldConfig": {
2024
+ "defaults": {
2025
+ "unit": "short"
2026
+ },
2027
+ "overrides": []
2028
+ }
2029
+ },
2030
+ {
2031
+ "type": "timeseries",
2032
+ "id": 13,
2033
+ "title": "Output tokens by model",
2034
+ "description": "Tokens per second by model, from genai_tokens_output_total.",
2035
+ "gridPos": {
2036
+ "h": 8,
2037
+ "w": 8,
2038
+ "x": 8,
2039
+ "y": 28
2040
+ },
2041
+ "datasource": {
2042
+ "type": "prometheus",
2043
+ "uid": "prometheus"
2044
+ },
2045
+ "targets": [
2046
+ {
2047
+ "datasource": {
2048
+ "type": "prometheus",
2049
+ "uid": "prometheus"
2050
+ },
2051
+ "refId": "A",
2052
+ "expr": "sum by (gen_ai_request_model) (rate(genai_tokens_output_total{gen_ai_request_model=~\"$model\"}[$__rate_interval]))",
2053
+ "legendFormat": "{{gen_ai_request_model}}"
2054
+ }
2055
+ ],
2056
+ "options": {
2057
+ "tooltip": {
2058
+ "mode": "multi"
2059
+ }
2060
+ },
2061
+ "fieldConfig": {
2062
+ "defaults": {
2063
+ "unit": "short"
2064
+ },
2065
+ "overrides": []
2066
+ }
2067
+ },
2068
+ {
2069
+ "type": "stat",
2070
+ "id": 14,
2071
+ "title": "Input tokens",
2072
+ "description": "Over the dashboard's time range.",
2073
+ "gridPos": {
2074
+ "h": 8,
2075
+ "w": 4,
2076
+ "x": 16,
2077
+ "y": 28
2078
+ },
2079
+ "datasource": {
2080
+ "type": "prometheus",
2081
+ "uid": "prometheus"
2082
+ },
2083
+ "targets": [
2084
+ {
2085
+ "datasource": {
2086
+ "type": "prometheus",
2087
+ "uid": "prometheus"
2088
+ },
2089
+ "refId": "A",
2090
+ "expr": "sum(increase(genai_tokens_input_total{gen_ai_request_model=~\"$model\"}[$__range]))",
2091
+ "legendFormat": "Input tokens",
2092
+ "instant": true,
2093
+ "range": false
2094
+ }
2095
+ ],
2096
+ "options": {
2097
+ "graphMode": "none",
2098
+ "colorMode": "none",
2099
+ "reduceOptions": {
2100
+ "calcs": [
2101
+ "lastNotNull"
2102
+ ]
2103
+ }
2104
+ },
2105
+ "fieldConfig": {
2106
+ "defaults": {
2107
+ "unit": "short",
2108
+ "decimals": 0
2109
+ },
2110
+ "overrides": []
2111
+ }
2112
+ },
2113
+ {
2114
+ "type": "stat",
2115
+ "id": 15,
2116
+ "title": "Output tokens",
2117
+ "description": "Over the dashboard's time range.",
2118
+ "gridPos": {
2119
+ "h": 8,
2120
+ "w": 4,
2121
+ "x": 20,
2122
+ "y": 28
2123
+ },
2124
+ "datasource": {
2125
+ "type": "prometheus",
2126
+ "uid": "prometheus"
2127
+ },
2128
+ "targets": [
2129
+ {
2130
+ "datasource": {
2131
+ "type": "prometheus",
2132
+ "uid": "prometheus"
2133
+ },
2134
+ "refId": "A",
2135
+ "expr": "sum(increase(genai_tokens_output_total{gen_ai_request_model=~\"$model\"}[$__range]))",
2136
+ "legendFormat": "Output tokens",
2137
+ "instant": true,
2138
+ "range": false
2139
+ }
2140
+ ],
2141
+ "options": {
2142
+ "graphMode": "none",
2143
+ "colorMode": "none",
2144
+ "reduceOptions": {
2145
+ "calcs": [
2146
+ "lastNotNull"
2147
+ ]
2148
+ }
2149
+ },
2150
+ "fieldConfig": {
2151
+ "defaults": {
2152
+ "unit": "short",
2153
+ "decimals": 0
2154
+ },
2155
+ "overrides": []
2156
+ }
2157
+ }
2158
+ ],
2159
+ "refresh": "1m",
2160
+ "schemaVersion": 42,
2161
+ "tags": [
2162
+ "genai",
2163
+ "agents"
2164
+ ],
2165
+ "templating": {
2166
+ "list": [
2167
+ {
2168
+ "type": "query",
2169
+ "name": "service",
2170
+ "label": "Service",
2171
+ "datasource": {
2172
+ "type": "prometheus",
2173
+ "uid": "prometheus"
2174
+ },
2175
+ "query": "label_values(genai_calls_total, service_name)",
2176
+ "definition": "label_values(genai_calls_total, service_name)",
2177
+ "refresh": 2,
2178
+ "sort": 1,
2179
+ "multi": true,
2180
+ "includeAll": true,
2181
+ "allValue": ".*",
2182
+ "options": []
2183
+ },
2184
+ {
2185
+ "type": "query",
2186
+ "name": "model",
2187
+ "label": "Model",
2188
+ "datasource": {
2189
+ "type": "prometheus",
2190
+ "uid": "prometheus"
2191
+ },
2192
+ "query": "label_values(genai_calls_total, gen_ai_request_model)",
2193
+ "definition": "label_values(genai_calls_total, gen_ai_request_model)",
2194
+ "refresh": 2,
2195
+ "sort": 1,
2196
+ "multi": true,
2197
+ "includeAll": true,
2198
+ "allValue": ".*",
2199
+ "options": []
2200
+ }
2201
+ ]
2202
+ },
2203
+ "time": {
2204
+ "from": "now-6h",
2205
+ "to": "now"
2206
+ },
2207
+ "timepicker": {},
2208
+ "timezone": "browser",
2209
+ "title": "Agents: models, tools and tokens",
2210
+ "uid": "genai-agents",
2211
+ "weekStart": ""
2212
+ }
2213
+
2214
+ ---
2215
+ apiVersion: v1
2216
+ kind: ConfigMap
2217
+ metadata:
2218
+ name: grafana-datasources
2219
+ namespace: observability
2220
+ labels:
2221
+ app.kubernetes.io/name: grafana
2222
+ app.kubernetes.io/component: dashboards
2223
+ grafana_datasource: '1'
2224
+ data:
2225
+ chant.yaml: |
2226
+ # Grafana datasource provisioning, generated by chant.
2227
+ apiVersion: 1
2228
+ datasources:
2229
+ - name: Loki
2230
+ type: loki
2231
+ uid: loki
2232
+ access: proxy
2233
+ url: http://loki:3100
2234
+ editable: false
2235
+ - name: Prometheus
2236
+ type: prometheus
2237
+ uid: prometheus
2238
+ access: proxy
2239
+ url: http://prometheus
2240
+ isDefault: true
2241
+ editable: false
2242
+ - name: Tempo
2243
+ type: tempo
2244
+ uid: tempo
2245
+ access: proxy
2246
+ url: http://tempo:3200
2247
+ jsonData:
2248
+ tracesToLogsV2:
2249
+ datasourceUid: loki
2250
+ filterByTraceID: true
2251
+ spanStartTimeShift: "-5m"
2252
+ spanEndTimeShift: 5m
2253
+ serviceMap:
2254
+ datasourceUid: prometheus
2255
+ editable: false
2256
+
2257
+ ---
2258
+ apiVersion: v1
2259
+ kind: ConfigMap
2260
+ metadata:
2261
+ name: grafana-dashboard-providers
2262
+ namespace: observability
2263
+ labels:
2264
+ app.kubernetes.io/name: grafana
2265
+ app.kubernetes.io/component: dashboards
2266
+ data:
2267
+ chant.yaml: |
2268
+ # Grafana dashboard provisioning, generated by chant.
2269
+ apiVersion: 1
2270
+ providers:
2271
+ - name: chant
2272
+ orgId: 1
2273
+ folder: ""
2274
+ type: file
2275
+ disableDeletion: false
2276
+ allowUiUpdates: false
2277
+ updateIntervalSeconds: 30
2278
+ options:
2279
+ path: /var/lib/grafana/dashboards
2280
+ foldersFromFilesStructure: true
2281
+
2282
+ ---
2283
+ apiVersion: v1
2284
+ kind: Service
2285
+ metadata:
2286
+ name: grafana
2287
+ namespace: observability
2288
+ labels:
2289
+ app.kubernetes.io/name: grafana
2290
+ app.kubernetes.io/component: dashboards
2291
+ spec:
2292
+ selector:
2293
+ app.kubernetes.io/name: grafana
2294
+ ports:
2295
+ - name: http
2296
+ port: 80
2297
+ targetPort: http
2298
+
2299
+ ---
2300
+ apiVersion: apps/v1
2301
+ kind: Deployment
2302
+ metadata:
2303
+ name: loki
2304
+ namespace: observability
2305
+ labels:
2306
+ app.kubernetes.io/name: loki
2307
+ app.kubernetes.io/component: logs
2308
+ spec:
2309
+ replicas: 1
2310
+ selector:
2311
+ matchLabels:
2312
+ app.kubernetes.io/name: loki
2313
+ template:
2314
+ metadata:
2315
+ labels:
2316
+ app.kubernetes.io/name: loki
2317
+ app.kubernetes.io/component: logs
2318
+ spec:
2319
+ securityContext:
2320
+ fsGroup: 10001
2321
+ containers:
2322
+ - name: loki
2323
+ image: grafana/loki:3.6.0
2324
+ imagePullPolicy: IfNotPresent
2325
+ args:
2326
+ - -config.file=/etc/loki/loki.yaml
2327
+ ports:
2328
+ - name: http
2329
+ containerPort: 3100
2330
+ readinessProbe:
2331
+ httpGet:
2332
+ path: /ready
2333
+ port: http
2334
+ initialDelaySeconds: 15
2335
+ periodSeconds: 5
2336
+ livenessProbe:
2337
+ tcpSocket:
2338
+ port: http
2339
+ initialDelaySeconds: 30
2340
+ periodSeconds: 10
2341
+ resources:
2342
+ requests:
2343
+ cpu: '100m'
2344
+ memory: '256Mi'
2345
+ limits:
2346
+ cpu: '1'
2347
+ memory: '1Gi'
2348
+ securityContext:
2349
+ runAsNonRoot: true
2350
+ runAsUser: 10001
2351
+ readOnlyRootFilesystem: true
2352
+ allowPrivilegeEscalation: false
2353
+ capabilities:
2354
+ drop:
2355
+ - ALL
2356
+ volumeMounts:
2357
+ - name: config
2358
+ mountPath: /etc/loki
2359
+ readOnly: true
2360
+ - name: data
2361
+ mountPath: /loki
2362
+ - name: tmp
2363
+ mountPath: /tmp
2364
+ volumes:
2365
+ - name: config
2366
+ configMap:
2367
+ name: loki-config
2368
+ - name: data
2369
+ emptyDir: {}
2370
+ - name: tmp
2371
+ emptyDir: {}
2372
+
2373
+ ---
2374
+ apiVersion: v1
2375
+ kind: ConfigMap
2376
+ metadata:
2377
+ name: loki-config
2378
+ namespace: observability
2379
+ labels:
2380
+ app.kubernetes.io/name: loki
2381
+ app.kubernetes.io/component: logs
2382
+ data:
2383
+ loki.yaml: |
2384
+ auth_enabled: false
2385
+ server:
2386
+ http_listen_port: 3100
2387
+ grpc_listen_port: 9096
2388
+ common:
2389
+ instance_addr: 127.0.0.1
2390
+ path_prefix: /loki
2391
+ storage:
2392
+ filesystem:
2393
+ chunks_directory: /loki/chunks
2394
+ rules_directory: /loki/rules
2395
+ replication_factor: 1
2396
+ ring:
2397
+ kvstore:
2398
+ store: inmemory
2399
+ schema_config:
2400
+ configs:
2401
+ - from: 2024-01-01
2402
+ store: tsdb
2403
+ object_store: filesystem
2404
+ schema: v13
2405
+ index:
2406
+ prefix: index_
2407
+ period: 24h
2408
+ limits_config:
2409
+ allow_structured_metadata: true
2410
+ analytics:
2411
+ reporting_enabled: false
2412
+
2413
+ ---
2414
+ apiVersion: v1
2415
+ kind: Service
2416
+ metadata:
2417
+ name: loki
2418
+ namespace: observability
2419
+ labels:
2420
+ app.kubernetes.io/name: loki
2421
+ app.kubernetes.io/component: logs
2422
+ spec:
2423
+ selector:
2424
+ app.kubernetes.io/name: loki
2425
+ ports:
2426
+ - name: http
2427
+ port: 3100
2428
+ targetPort: http
2429
+
2430
+ ---
2431
+ apiVersion: apps/v1
2432
+ kind: Deployment
2433
+ metadata:
2434
+ name: alertmanager
2435
+ namespace: observability
2436
+ labels:
2437
+ app.kubernetes.io/name: alertmanager
2438
+ app.kubernetes.io/managed-by: chant
2439
+ app.kubernetes.io/component: server
2440
+ spec:
2441
+ replicas: 1
2442
+ selector:
2443
+ matchLabels:
2444
+ app.kubernetes.io/name: alertmanager
2445
+ template:
2446
+ metadata:
2447
+ labels:
2448
+ app.kubernetes.io/name: alertmanager
2449
+ spec:
2450
+ containers:
2451
+ - name: alertmanager
2452
+ image: prom/alertmanager:v0.34.1
2453
+ ports:
2454
+ - containerPort: 9093
2455
+ name: http
2456
+ resources:
2457
+ limits:
2458
+ cpu: '500m'
2459
+ memory: '256Mi'
2460
+ requests:
2461
+ cpu: '100m'
2462
+ memory: '128Mi'
2463
+ volumeMounts:
2464
+ - name: config
2465
+ mountPath: /etc/alertmanager
2466
+ readOnly: true
2467
+ securityContext:
2468
+ runAsNonRoot: true
2469
+ runAsUser: 65534
2470
+ allowPrivilegeEscalation: false
2471
+ capabilities:
2472
+ drop:
2473
+ - ALL
2474
+ volumes:
2475
+ - name: config
2476
+ configMap:
2477
+ name: alertmanager-config
2478
+
2479
+ ---
2480
+ apiVersion: v1
2481
+ kind: Service
2482
+ metadata:
2483
+ name: alertmanager
2484
+ namespace: observability
2485
+ labels:
2486
+ app.kubernetes.io/name: alertmanager
2487
+ app.kubernetes.io/managed-by: chant
2488
+ app.kubernetes.io/component: server
2489
+ spec:
2490
+ selector:
2491
+ app.kubernetes.io/name: alertmanager
2492
+ ports:
2493
+ - port: 80
2494
+ targetPort: 9093
2495
+ protocol: TCP
2496
+ name: http
2497
+ type: ClusterIP
2498
+
2499
+ ---
2500
+ apiVersion: v1
2501
+ kind: ConfigMap
2502
+ metadata:
2503
+ name: alertmanager-config
2504
+ namespace: observability
2505
+ labels:
2506
+ app.kubernetes.io/name: alertmanager
2507
+ app.kubernetes.io/managed-by: chant
2508
+ app.kubernetes.io/component: config
2509
+ data:
2510
+ alertmanager.yml: |
2511
+ route:
2512
+ receiver: default
2513
+ group_by:
2514
+ - alertname
2515
+ - slo
2516
+ group_wait: 10s
2517
+ group_interval: 1m
2518
+ routes:
2519
+ - receiver: oncall
2520
+ matchers:
2521
+ - severity="page"
2522
+ repeat_interval: 1h
2523
+ - receiver: tickets
2524
+ matchers:
2525
+ - severity="ticket"
2526
+ repeat_interval: 12h
2527
+ inhibit_rules:
2528
+ - source_matchers:
2529
+ - severity="page"
2530
+ target_matchers:
2531
+ - severity="ticket"
2532
+ equal:
2533
+ - slo
2534
+ receivers:
2535
+ - name: default
2536
+ - name: oncall
2537
+ - name: tickets
2538
+
2539
+ ---
2540
+ apiVersion: apps/v1
2541
+ kind: Deployment
2542
+ metadata:
2543
+ name: prometheus
2544
+ namespace: observability
2545
+ labels:
2546
+ app.kubernetes.io/name: prometheus
2547
+ app.kubernetes.io/managed-by: chant
2548
+ app.kubernetes.io/component: server
2549
+ spec:
2550
+ replicas: 1
2551
+ selector:
2552
+ matchLabels:
2553
+ app.kubernetes.io/name: prometheus
2554
+ template:
2555
+ metadata:
2556
+ labels:
2557
+ app.kubernetes.io/name: prometheus
2558
+ spec:
2559
+ containers:
2560
+ - name: prometheus
2561
+ image: prom/prometheus:v3.15.0
2562
+ ports:
2563
+ - containerPort: 9090
2564
+ name: http
2565
+ resources:
2566
+ limits:
2567
+ cpu: '500m'
2568
+ memory: '512Mi'
2569
+ requests:
2570
+ cpu: '100m'
2571
+ memory: '128Mi'
2572
+ volumeMounts:
2573
+ - name: config
2574
+ mountPath: /etc/prometheus
2575
+ readOnly: true
2576
+ securityContext:
2577
+ runAsNonRoot: true
2578
+ runAsUser: 65534
2579
+ allowPrivilegeEscalation: false
2580
+ capabilities:
2581
+ drop:
2582
+ - ALL
2583
+ volumes:
2584
+ - name: config
2585
+ configMap:
2586
+ name: prometheus-config
2587
+
2588
+ ---
2589
+ apiVersion: v1
2590
+ kind: Service
2591
+ metadata:
2592
+ name: prometheus
2593
+ namespace: observability
2594
+ labels:
2595
+ app.kubernetes.io/name: prometheus
2596
+ app.kubernetes.io/managed-by: chant
2597
+ app.kubernetes.io/component: server
2598
+ spec:
2599
+ selector:
2600
+ app.kubernetes.io/name: prometheus
2601
+ ports:
2602
+ - port: 80
2603
+ targetPort: 9090
2604
+ protocol: TCP
2605
+ name: http
2606
+ type: ClusterIP
2607
+
2608
+ ---
2609
+ apiVersion: v1
2610
+ kind: ConfigMap
2611
+ metadata:
2612
+ name: prometheus-config
2613
+ namespace: observability
2614
+ labels:
2615
+ app.kubernetes.io/name: prometheus
2616
+ app.kubernetes.io/managed-by: chant
2617
+ app.kubernetes.io/component: config
2618
+ data:
2619
+ prometheus.yml: |
2620
+ global:
2621
+ scrape_interval: 15s
2622
+ evaluation_interval: 15s
2623
+ rule_files:
2624
+ - /etc/prometheus/rules.yml
2625
+ alerting:
2626
+ alertmanagers:
2627
+ - static_configs:
2628
+ - targets: ["alertmanager:80"]
2629
+ scrape_configs:
2630
+ - job_name: otel-gateway
2631
+ dns_sd_configs:
2632
+ - names: ["otel-gateway-headless.observability.svc.cluster.local"]
2633
+ type: A
2634
+ port: 8889
2635
+ refresh_interval: 15s
2636
+ - job_name: prometheus
2637
+ static_configs:
2638
+ - targets: ["localhost:9090"]
2639
+ rules.yml: |
2640
+ groups:
2641
+ - name: slo-support-agent-runs
2642
+ interval: 15s
2643
+ labels:
2644
+ team: agents
2645
+ rules:
2646
+ - record: slo:sli_error:ratio_rate5m
2647
+ expr: |-
2648
+ (sum(rate(traces_span_metrics_calls_total{service_name="support-agent",span_name="invoke_agent support",status_code="STATUS_CODE_ERROR"}[5m])))
2649
+ /
2650
+ (sum(rate(traces_span_metrics_calls_total{service_name="support-agent",span_name="invoke_agent support"}[5m])))
2651
+ labels:
2652
+ slo: support-agent-runs
2653
+ - record: slo:sli_error:ratio_rate30m
2654
+ expr: |-
2655
+ (sum(rate(traces_span_metrics_calls_total{service_name="support-agent",span_name="invoke_agent support",status_code="STATUS_CODE_ERROR"}[30m])))
2656
+ /
2657
+ (sum(rate(traces_span_metrics_calls_total{service_name="support-agent",span_name="invoke_agent support"}[30m])))
2658
+ labels:
2659
+ slo: support-agent-runs
2660
+ - record: slo:sli_error:ratio_rate1h
2661
+ expr: |-
2662
+ (sum(rate(traces_span_metrics_calls_total{service_name="support-agent",span_name="invoke_agent support",status_code="STATUS_CODE_ERROR"}[1h])))
2663
+ /
2664
+ (sum(rate(traces_span_metrics_calls_total{service_name="support-agent",span_name="invoke_agent support"}[1h])))
2665
+ labels:
2666
+ slo: support-agent-runs
2667
+ - record: slo:sli_error:ratio_rate2h
2668
+ expr: |-
2669
+ (sum(rate(traces_span_metrics_calls_total{service_name="support-agent",span_name="invoke_agent support",status_code="STATUS_CODE_ERROR"}[2h])))
2670
+ /
2671
+ (sum(rate(traces_span_metrics_calls_total{service_name="support-agent",span_name="invoke_agent support"}[2h])))
2672
+ labels:
2673
+ slo: support-agent-runs
2674
+ - record: slo:sli_error:ratio_rate6h
2675
+ expr: |-
2676
+ (sum(rate(traces_span_metrics_calls_total{service_name="support-agent",span_name="invoke_agent support",status_code="STATUS_CODE_ERROR"}[6h])))
2677
+ /
2678
+ (sum(rate(traces_span_metrics_calls_total{service_name="support-agent",span_name="invoke_agent support"}[6h])))
2679
+ labels:
2680
+ slo: support-agent-runs
2681
+ - record: slo:sli_error:ratio_rate1d
2682
+ expr: |-
2683
+ (sum(rate(traces_span_metrics_calls_total{service_name="support-agent",span_name="invoke_agent support",status_code="STATUS_CODE_ERROR"}[1d])))
2684
+ /
2685
+ (sum(rate(traces_span_metrics_calls_total{service_name="support-agent",span_name="invoke_agent support"}[1d])))
2686
+ labels:
2687
+ slo: support-agent-runs
2688
+ - record: slo:sli_error:ratio_rate3d
2689
+ expr: |-
2690
+ (sum(rate(traces_span_metrics_calls_total{service_name="support-agent",span_name="invoke_agent support",status_code="STATUS_CODE_ERROR"}[3d])))
2691
+ /
2692
+ (sum(rate(traces_span_metrics_calls_total{service_name="support-agent",span_name="invoke_agent support"}[3d])))
2693
+ labels:
2694
+ slo: support-agent-runs
2695
+ - record: slo:sli_error:ratio_rate28d
2696
+ expr: avg_over_time(slo:sli_error:ratio_rate5m{slo="support-agent-runs"}[28d])
2697
+ labels:
2698
+ slo: support-agent-runs
2699
+ - record: slo:objective:ratio
2700
+ expr: vector(0.99)
2701
+ labels:
2702
+ slo: support-agent-runs
2703
+ - record: slo:error_budget:remaining
2704
+ expr: 1 - (slo:sli_error:ratio_rate28d{slo="support-agent-runs"} / 0.01)
2705
+ labels:
2706
+ slo: support-agent-runs
2707
+ - alert: ErrorBudgetBurn
2708
+ expr: |-
2709
+ (
2710
+ slo:sli_error:ratio_rate1h{slo="support-agent-runs"} > (13.44 * 0.01)
2711
+ )
2712
+ and
2713
+ (
2714
+ slo:sli_error:ratio_rate5m{slo="support-agent-runs"} > (13.44 * 0.01)
2715
+ )
2716
+ labels:
2717
+ slo: support-agent-runs
2718
+ severity: page
2719
+ long_window: 1h
2720
+ short_window: 5m
2721
+ annotations:
2722
+ summary: SLO support-agent-runs is burning its error budget more than 13.44x too fast (1h and 5m windows)
2723
+ description: Support agent runs end without an error span. The error ratio over both the last 1h and the last 5m is above 0.1344 (13.44 times the 0.01 budget of a 0.99 objective). At that rate the 28d error budget is gone in 2d2h.
2724
+ - alert: ErrorBudgetBurn
2725
+ expr: |-
2726
+ (
2727
+ slo:sli_error:ratio_rate6h{slo="support-agent-runs"} > (5.6 * 0.01)
2728
+ )
2729
+ and
2730
+ (
2731
+ slo:sli_error:ratio_rate30m{slo="support-agent-runs"} > (5.6 * 0.01)
2732
+ )
2733
+ labels:
2734
+ slo: support-agent-runs
2735
+ severity: page
2736
+ long_window: 6h
2737
+ short_window: 30m
2738
+ annotations:
2739
+ summary: SLO support-agent-runs is burning its error budget more than 5.6x too fast (6h and 30m windows)
2740
+ description: Support agent runs end without an error span. The error ratio over both the last 6h and the last 30m is above 0.056 (5.6 times the 0.01 budget of a 0.99 objective). At that rate the 28d error budget is gone in 5d.
2741
+ - alert: ErrorBudgetBurn
2742
+ expr: |-
2743
+ (
2744
+ slo:sli_error:ratio_rate1d{slo="support-agent-runs"} > (2.8 * 0.01)
2745
+ )
2746
+ and
2747
+ (
2748
+ slo:sli_error:ratio_rate2h{slo="support-agent-runs"} > (2.8 * 0.01)
2749
+ )
2750
+ labels:
2751
+ slo: support-agent-runs
2752
+ severity: ticket
2753
+ long_window: 1d
2754
+ short_window: 2h
2755
+ annotations:
2756
+ summary: SLO support-agent-runs is burning its error budget more than 2.8x too fast (1d and 2h windows)
2757
+ description: Support agent runs end without an error span. The error ratio over both the last 1d and the last 2h is above 0.028 (2.8 times the 0.01 budget of a 0.99 objective). At that rate the 28d error budget is gone in 1w3d.
2758
+ - alert: ErrorBudgetBurn
2759
+ expr: |-
2760
+ (
2761
+ slo:sli_error:ratio_rate3d{slo="support-agent-runs"} > (0.933333 * 0.01)
2762
+ )
2763
+ and
2764
+ (
2765
+ slo:sli_error:ratio_rate6h{slo="support-agent-runs"} > (0.933333 * 0.01)
2766
+ )
2767
+ labels:
2768
+ slo: support-agent-runs
2769
+ severity: ticket
2770
+ long_window: 3d
2771
+ short_window: 6h
2772
+ annotations:
2773
+ summary: SLO support-agent-runs is burning its error budget more than 0.933333x too fast (3d and 6h windows)
2774
+ description: Support agent runs end without an error span. The error ratio over both the last 3d and the last 6h is above 0.00933333 (0.933333 times the 0.01 budget of a 0.99 objective). At that rate the 28d error budget is gone in 4w2d.
2775
+
2776
+ ---
2777
+ apiVersion: apps/v1
2778
+ kind: Deployment
2779
+ metadata:
2780
+ name: tempo
2781
+ namespace: observability
2782
+ labels:
2783
+ app.kubernetes.io/name: tempo
2784
+ app.kubernetes.io/component: traces
2785
+ spec:
2786
+ replicas: 1
2787
+ selector:
2788
+ matchLabels:
2789
+ app.kubernetes.io/name: tempo
2790
+ template:
2791
+ metadata:
2792
+ labels:
2793
+ app.kubernetes.io/name: tempo
2794
+ app.kubernetes.io/component: traces
2795
+ spec:
2796
+ securityContext:
2797
+ fsGroup: 10001
2798
+ containers:
2799
+ - name: tempo
2800
+ image: grafana/tempo:2.9.0
2801
+ imagePullPolicy: IfNotPresent
2802
+ args:
2803
+ - -config.file=/etc/tempo/tempo.yaml
2804
+ ports:
2805
+ - name: http
2806
+ containerPort: 3200
2807
+ - name: otlp-grpc
2808
+ containerPort: 4317
2809
+ readinessProbe:
2810
+ httpGet:
2811
+ path: /ready
2812
+ port: http
2813
+ initialDelaySeconds: 10
2814
+ periodSeconds: 5
2815
+ livenessProbe:
2816
+ tcpSocket:
2817
+ port: http
2818
+ initialDelaySeconds: 30
2819
+ periodSeconds: 10
2820
+ resources:
2821
+ requests:
2822
+ cpu: '100m'
2823
+ memory: '256Mi'
2824
+ limits:
2825
+ cpu: '1'
2826
+ memory: '1Gi'
2827
+ securityContext:
2828
+ runAsNonRoot: true
2829
+ runAsUser: 10001
2830
+ readOnlyRootFilesystem: true
2831
+ allowPrivilegeEscalation: false
2832
+ capabilities:
2833
+ drop:
2834
+ - ALL
2835
+ volumeMounts:
2836
+ - name: config
2837
+ mountPath: /etc/tempo
2838
+ readOnly: true
2839
+ - name: data
2840
+ mountPath: /var/tempo
2841
+ - name: tmp
2842
+ mountPath: /tmp
2843
+ volumes:
2844
+ - name: config
2845
+ configMap:
2846
+ name: tempo-config
2847
+ - name: data
2848
+ emptyDir: {}
2849
+ - name: tmp
2850
+ emptyDir: {}
2851
+
2852
+ ---
2853
+ apiVersion: v1
2854
+ kind: ConfigMap
2855
+ metadata:
2856
+ name: tempo-config
2857
+ namespace: observability
2858
+ labels:
2859
+ app.kubernetes.io/name: tempo
2860
+ app.kubernetes.io/component: traces
2861
+ data:
2862
+ tempo.yaml: |
2863
+ stream_over_http_enabled: true
2864
+ server:
2865
+ http_listen_port: 3200
2866
+ distributor:
2867
+ receivers:
2868
+ otlp:
2869
+ protocols:
2870
+ grpc:
2871
+ endpoint: 0.0.0.0:4317
2872
+ ingester:
2873
+ max_block_duration: 5m
2874
+ compactor:
2875
+ compaction:
2876
+ block_retention: 24h
2877
+ storage:
2878
+ trace:
2879
+ backend: local
2880
+ wal:
2881
+ path: /var/tempo/wal
2882
+ local:
2883
+ path: /var/tempo/blocks
2884
+ usage_report:
2885
+ reporting_enabled: false
2886
+
2887
+ ---
2888
+ apiVersion: v1
2889
+ kind: Service
2890
+ metadata:
2891
+ name: tempo
2892
+ namespace: observability
2893
+ labels:
2894
+ app.kubernetes.io/name: tempo
2895
+ app.kubernetes.io/component: traces
2896
+ spec:
2897
+ selector:
2898
+ app.kubernetes.io/name: tempo
2899
+ ports:
2900
+ - name: http
2901
+ port: 3200
2902
+ targetPort: http
2903
+ - name: otlp-grpc
2904
+ port: 4317
2905
+ targetPort: otlp-grpc