devopsiq 0.1.0__py3-none-any.whl

This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
tools/kubernetes.py ADDED
@@ -0,0 +1,464 @@
1
+ """Read-only Kubernetes investigation tools (Phase 3).
2
+
3
+ Three kubectl-backed tools: pod status (the CrashLoopBackOff evidence
4
+ source), pod logs, and deployment/rollout status. They follow the same
5
+ contract and safety model as tools/preflight.py but talk to a cluster via
6
+ `kubectl` — no extra Python dependency, and parity with what ops teams
7
+ actually run.
8
+
9
+ Safety model (defense in depth):
10
+ - Only the `get` and `logs` kubectl verbs exist, hard-coded in the argv
11
+ templates below. There is no delete, restart, edit, apply, scale, exec or
12
+ create path — and never a `sh -c`, so nothing is ever parsed by a shell.
13
+ - Object names (pod / deployment / namespace) are validated against the
14
+ Kubernetes naming rules before touching kubectl; this blocks flag
15
+ injection (names starting with "-") and garbage input. Never trust the
16
+ model's arguments.
17
+ - `--request-timeout` plus a subprocess timeout bound slow/hung clusters.
18
+ - Output is real kubectl JSON/text, truncated to the shared cap. Python does
19
+ NOT parse the pod/deployment spec (stable across kubectl versions and all
20
+ cluster states); the model reads the raw output, which is exactly the
21
+ evidence an engineer would see.
22
+ - kubectl must be installed and configured (KUBECONFIG / default context)
23
+ on the host. If it is missing or the cluster is unreachable, the exact
24
+ error is returned — nothing is ever invented.
25
+ """
26
+
27
+ import re
28
+
29
+ from tools.base import Tool, ToolError, read_command_output
30
+ from tools.registry import register
31
+
32
+ # Kubernetes object names: lowercase letters, digits, '-' or '.', DNS-style,
33
+ # at most 253 characters, starting and ending with an alphanumeric.
34
+ _NAME_RE = re.compile(r"[a-z0-9](?:[-a-z0-9.]{0,251}[a-z0-9])?")
35
+
36
+ # Bound cluster queries so a hanging API server (or `kubectl logs`
37
+ # following a stream) can never stall a turn.
38
+ _TIMEOUT_S = 15
39
+
40
+
41
+ def _check_name(value, kind: str) -> str:
42
+ if (
43
+ not isinstance(value, str)
44
+ or len(value) > 253
45
+ or not _NAME_RE.fullmatch(value)
46
+ ):
47
+ raise ToolError(
48
+ f"invalid Kubernetes {kind} name {value!r}: expected lowercase "
49
+ "letters, digits, '-' or '.', max 253 characters"
50
+ )
51
+ return value
52
+
53
+
54
+ def _namespace(args: dict) -> str:
55
+ return _check_name(args.get("namespace") or "default", "namespace")
56
+
57
+
58
+ def _k8s_pod_status(args: dict) -> str:
59
+ pod = _check_name(args.get("pod"), "pod")
60
+ namespace = _namespace(args)
61
+ return read_command_output(
62
+ ("kubectl", "get", "pod", pod, "-n", namespace, "-o", "json",
63
+ "--request-timeout=10"),
64
+ timeout=_TIMEOUT_S,
65
+ )
66
+
67
+
68
+ def _k8s_pod_logs(args: dict) -> str:
69
+ pod = _check_name(args.get("pod"), "pod")
70
+ namespace = _namespace(args)
71
+ try:
72
+ lines = int(args.get("lines", 100))
73
+ except (TypeError, ValueError):
74
+ raise ToolError("lines must be an integer between 1 and 500")
75
+ if not 1 <= lines <= 500:
76
+ raise ToolError("lines must be between 1 and 500")
77
+ return read_command_output(
78
+ ("kubectl", "logs", pod, "-n", namespace, "--tail", str(lines),
79
+ "--request-timeout=10"),
80
+ timeout=_TIMEOUT_S,
81
+ )
82
+
83
+
84
+ def _k8s_deployment_status(args: dict) -> str:
85
+ deployment = _check_name(args.get("deployment"), "deployment")
86
+ namespace = _namespace(args)
87
+ return read_command_output(
88
+ ("kubectl", "get", "deployment", deployment, "-n", namespace, "-o",
89
+ "json", "--request-timeout=10"),
90
+ timeout=_TIMEOUT_S,
91
+ )
92
+
93
+
94
+ # Human-readable hint that is NOT part of the docs — helps readme readers —
95
+ # but the schema `description` is what the model actually sees.
96
+ K8S_POD_STATUS = Tool(
97
+ name="k8s_pod_status",
98
+ description=(
99
+ "Fetch a Kubernetes pod (by name, in a namespace) as JSON, using "
100
+ "kubectl's pod liveness API. Key fields: 'status.phase' (Running/..."
101
+ "), 'status.containerStatuses[].restartCount' (climbing count on "
102
+ "repeated crashes), and 'status.conditions[]' diagnostics (e.g. "
103
+ "reason 'ContainersNotReady' with a message naming unready "
104
+ "containers) — the evidence for CrashLoopBackOff investigations. "
105
+ "Requires kubectl configured against a cluster. Read-only."
106
+ ),
107
+ parameters={
108
+ "type": "object",
109
+ "properties": {
110
+ "pod": {"type": "string", "description": "Name of the pod."},
111
+ "namespace": {
112
+ "type": "string",
113
+ "description": "Namespace of the pod (default: \"default\").",
114
+ },
115
+ },
116
+ "required": ["pod"],
117
+ "additionalProperties": False,
118
+ },
119
+ executor=_k8s_pod_status,
120
+ )
121
+
122
+ K8S_POD_LOGS = Tool(
123
+ name="k8s_pod_logs",
124
+ description=(
125
+ "Fetch the tail of a pod's logs as plain text. Use with k8s_pod_status "
126
+ "to see the crash/backoff error messages. Requires kubectl configured "
127
+ "against a cluster. Read-only."
128
+ ),
129
+ parameters={
130
+ "type": "object",
131
+ "properties": {
132
+ "pod": {"type": "string", "description": "Name of the pod."},
133
+ "namespace": {
134
+ "type": "string",
135
+ "description": "Namespace of the pod (default: \"default\").",
136
+ },
137
+ "lines": {
138
+ "type": "integer",
139
+ "minimum": 1,
140
+ "maximum": 500,
141
+ "default": 100,
142
+ "description": "How many lines to tail (1–500).",
143
+ },
144
+ },
145
+ "required": ["pod"],
146
+ "additionalProperties": False,
147
+ },
148
+ executor=_k8s_pod_logs,
149
+ )
150
+
151
+ K8S_DEPLOYMENT_STATUS = Tool(
152
+ name="k8s_deployment_status",
153
+ description=(
154
+ "Fetch a Kubernetes Deployment (by name, in a namespace) as JSON, "
155
+ "including metadata and rollout status. Use to assess whether a "
156
+ "rollout completed, is degraded, or failed. Requires kubectl "
157
+ "configured against a cluster. Read-only."
158
+ ),
159
+ parameters={
160
+ "type": "object",
161
+ "properties": {
162
+ "deployment": {"type": "string", "description": "Name of the Deployment."},
163
+ "namespace": {
164
+ "type": "string",
165
+ "description": "Namespace of the Deployment (default: \"default\").",
166
+ },
167
+ },
168
+ "required": ["deployment"],
169
+ "additionalProperties": False,
170
+ },
171
+ executor=_k8s_deployment_status,
172
+ )
173
+
174
+ def _k8s_events(args: dict) -> str:
175
+ namespace = _namespace(args)
176
+ argv = [
177
+ "kubectl", "get", "events",
178
+ "-n", namespace,
179
+ "--sort-by=.lastTimestamp",
180
+ "-o", "wide",
181
+ ]
182
+ # Optional: filter to events about one object (validated name only).
183
+ involving = args.get("involving")
184
+ if involving is not None:
185
+ argv += ["--field-selector", f"involvedObject.name={_check_name(involving, 'object')}"]
186
+ argv += ["--request-timeout=10"]
187
+ return read_command_output(tuple(argv), timeout=_TIMEOUT_S)
188
+
189
+
190
+ def _k8s_nodes(args: dict) -> str:
191
+ return read_command_output(
192
+ ("kubectl", "get", "nodes", "-o", "json", "--request-timeout=10"),
193
+ timeout=_TIMEOUT_S,
194
+ )
195
+
196
+
197
+ def _k8s_services(args: dict) -> str:
198
+ namespace = _namespace(args)
199
+ return read_command_output(
200
+ ("kubectl", "get", "services", "-n", namespace, "-o", "json",
201
+ "--request-timeout=10"),
202
+ timeout=_TIMEOUT_S,
203
+ )
204
+
205
+
206
+ K8S_EVENTS = Tool(
207
+ name="k8s_events",
208
+ description=(
209
+ "Fetch recent cluster Events in a namespace, newest last (kubectl get "
210
+ "events, wide): Warning/type, reason, message, object, and timestamps "
211
+ "for pods, deployments etc. The record of WHAT happened — use it when "
212
+ "pod/deployment conditions don't explain a failure. Read-only."
213
+ ),
214
+ parameters={
215
+ "type": "object",
216
+ "properties": {
217
+ "namespace": {
218
+ "type": "string",
219
+ "description": "Namespace to read events from (default: \"default\").",
220
+ },
221
+ "involving": {
222
+ "type": "string",
223
+ "description": "Optional object name to filter events to "
224
+ "(e.g. a pod name).",
225
+ },
226
+ },
227
+ "additionalProperties": False,
228
+ },
229
+ executor=_k8s_events,
230
+ )
231
+
232
+ K8S_NODES = Tool(
233
+ name="k8s_nodes",
234
+ description=(
235
+ "Fetch all cluster nodes as JSON: status conditions (Ready/MemoryPressure/"
236
+ "DiskPressure), roles, kubelet version, taints. Use for node-level "
237
+ "problems — a node NotReady, scheduling issues, capacity. Read-only."
238
+ ),
239
+ parameters={
240
+ "type": "object",
241
+ "properties": {},
242
+ "additionalProperties": False,
243
+ },
244
+ executor=_k8s_nodes,
245
+ )
246
+
247
+ K8S_SERVICES = Tool(
248
+ name="k8s_services",
249
+ description=(
250
+ "Fetch Services in a namespace as JSON: type (ClusterIP/LoadBalancer/"
251
+ "NodePort), cluster IP, ports, selectors. Use for 'why can't I reach "
252
+ "this service' / exposure problems. Read-only."
253
+ ),
254
+ parameters={
255
+ "type": "object",
256
+ "properties": {
257
+ "namespace": {
258
+ "type": "string",
259
+ "description": "Namespace of the Services (default: \"default\").",
260
+ },
261
+ },
262
+ "additionalProperties": False,
263
+ },
264
+ executor=_k8s_services,
265
+ )
266
+
267
+ # --- Phase 8: cluster depth — pod listing, resource usage, autoscaling,
268
+ # storage, and contexts. Same contract: fixed argv, validated names, get/top
269
+ # verbs only (never `config use-context`, which would mutate kubeconfig). ----
270
+
271
+
272
+ def _k8s_pods(args: dict) -> str:
273
+ namespace = _namespace(args)
274
+ return read_command_output(
275
+ ("kubectl", "get", "pods", "-n", namespace, "-o", "wide",
276
+ "--request-timeout=10"),
277
+ timeout=_TIMEOUT_S,
278
+ )
279
+
280
+
281
+ def _k8s_top_pods(args: dict) -> str:
282
+ namespace = _namespace(args)
283
+ argv = ["kubectl", "top", "pods", "-n", namespace]
284
+ sort_by = args.get("sort_by")
285
+ if sort_by is not None:
286
+ if sort_by not in ("cpu", "memory"):
287
+ raise ToolError("sort_by must be 'cpu' or 'memory'")
288
+ argv.append("--sort-by=" + sort_by)
289
+ argv.append("--request-timeout=10")
290
+ return read_command_output(tuple(argv), timeout=_TIMEOUT_S)
291
+
292
+
293
+ def _k8s_top_nodes(args: dict) -> str:
294
+ return read_command_output(
295
+ ("kubectl", "top", "nodes", "--request-timeout=10"),
296
+ timeout=_TIMEOUT_S,
297
+ )
298
+
299
+
300
+ def _k8s_hpa(args: dict) -> str:
301
+ namespace = _namespace(args)
302
+ name = args.get("name")
303
+ if name is not None:
304
+ _check_name(name, "hpa")
305
+ argv = ("kubectl", "get", "hpa", name, "-n", namespace, "-o", "json",
306
+ "--request-timeout=10")
307
+ else:
308
+ argv = ("kubectl", "get", "hpa", "-n", namespace, "-o", "json",
309
+ "--request-timeout=10")
310
+ return read_command_output(argv, timeout=_TIMEOUT_S)
311
+
312
+
313
+ def _k8s_pvc(args: dict) -> str:
314
+ namespace = _namespace(args)
315
+ name = args.get("name")
316
+ if name is not None:
317
+ _check_name(name, "pvc")
318
+ argv = ("kubectl", "get", "pvc", name, "-n", namespace, "-o", "json",
319
+ "--request-timeout=10")
320
+ else:
321
+ argv = ("kubectl", "get", "pvc", "-n", namespace, "-o", "json",
322
+ "--request-timeout=10")
323
+ return read_command_output(argv, timeout=_TIMEOUT_S)
324
+
325
+
326
+ def _k8s_contexts(args: dict) -> str:
327
+ # Local kubeconfig read — no cluster API, so no request timeout needed.
328
+ return read_command_output(("kubectl", "config", "get-contexts"))
329
+
330
+
331
+ K8S_PODS = Tool(
332
+ name="k8s_pods",
333
+ description=(
334
+ "List Pods in a namespace (kubectl get pods -o wide): name, ready "
335
+ "containers, status, restarts, age, and the node each runs on. The "
336
+ "first tool for 'what is running / what looks unhealthy' in a "
337
+ "namespace — follow up on interesting pods with k8s_pod_status. "
338
+ "Read-only."
339
+ ),
340
+ parameters={
341
+ "type": "object",
342
+ "properties": {
343
+ "namespace": {
344
+ "type": "string",
345
+ "description": "Namespace to list (default: \"default\").",
346
+ },
347
+ },
348
+ "additionalProperties": False,
349
+ },
350
+ executor=_k8s_pods,
351
+ )
352
+
353
+ K8S_TOP_PODS = Tool(
354
+ name="k8s_top_pods",
355
+ description=(
356
+ "Resource usage of Pods in a namespace (kubectl top pods): CPU and "
357
+ "memory per pod. Use for 'which pod burns CPU/RAM', capacity checks. "
358
+ "Requires metrics-server in the cluster; if absent, kubectl's exact "
359
+ "error is returned. Read-only."
360
+ ),
361
+ parameters={
362
+ "type": "object",
363
+ "properties": {
364
+ "namespace": {
365
+ "type": "string",
366
+ "description": "Namespace to measure (default: \"default\").",
367
+ },
368
+ "sort_by": {
369
+ "type": "string",
370
+ "enum": ["cpu", "memory"],
371
+ "description": "Optional sort column.",
372
+ },
373
+ },
374
+ "additionalProperties": False,
375
+ },
376
+ executor=_k8s_top_pods,
377
+ )
378
+
379
+ K8S_TOP_NODES = Tool(
380
+ name="k8s_top_nodes",
381
+ description=(
382
+ "Resource usage of all cluster nodes (kubectl top nodes): CPU and "
383
+ "memory per node, for node-level capacity and pressure questions. "
384
+ "Requires metrics-server; if absent, kubectl's exact error is "
385
+ "returned. Read-only."
386
+ ),
387
+ parameters={"type": "object", "properties": {}, "additionalProperties": False},
388
+ executor=_k8s_top_nodes,
389
+ )
390
+
391
+ K8S_HPA = Tool(
392
+ name="k8s_hpa",
393
+ description=(
394
+ "Fetch HorizontalPodAutoscalers in a namespace as JSON (kubectl get "
395
+ "hpa): current/target utilization, min/max replicas, and the scale "
396
+ "target. Use for autoscaling problems — a workload pinned at max "
397
+ "replicas or not scaling at all. Pass 'name' for one HPA. Read-only."
398
+ ),
399
+ parameters={
400
+ "type": "object",
401
+ "properties": {
402
+ "name": {
403
+ "type": "string",
404
+ "description": "Optional HPA name (omit to list all).",
405
+ },
406
+ "namespace": {
407
+ "type": "string",
408
+ "description": "Namespace (default: \"default\").",
409
+ },
410
+ },
411
+ "additionalProperties": False,
412
+ },
413
+ executor=_k8s_hpa,
414
+ )
415
+
416
+ K8S_PVC = Tool(
417
+ name="k8s_pvc",
418
+ description=(
419
+ "Fetch PersistentVolumeClaims in a namespace as JSON (kubectl get "
420
+ "pvc): phase (Bound/Pending/Lost), storage class, capacity, volume "
421
+ "name. Use for 'why is my volume Pending' / storage debugging. Pass "
422
+ "'name' for one PVC. Read-only."
423
+ ),
424
+ parameters={
425
+ "type": "object",
426
+ "properties": {
427
+ "name": {
428
+ "type": "string",
429
+ "description": "Optional PVC name (omit to list all).",
430
+ },
431
+ "namespace": {
432
+ "type": "string",
433
+ "description": "Namespace (default: \"default\").",
434
+ },
435
+ },
436
+ "additionalProperties": False,
437
+ },
438
+ executor=_k8s_pvc,
439
+ )
440
+
441
+ K8S_CONTEXTS = Tool(
442
+ name="k8s_contexts",
443
+ description=(
444
+ "List the kubeconfig contexts on this host (kubectl config "
445
+ "get-contexts), with the current context marked. Use to see WHICH "
446
+ "cluster the other k8s tools are pointed at. Listing only — the "
447
+ "agent can never switch contexts. Read-only."
448
+ ),
449
+ parameters={"type": "object", "properties": {}, "additionalProperties": False},
450
+ executor=_k8s_contexts,
451
+ )
452
+
453
+ register(K8S_POD_STATUS)
454
+ register(K8S_POD_LOGS)
455
+ register(K8S_DEPLOYMENT_STATUS)
456
+ register(K8S_EVENTS)
457
+ register(K8S_NODES)
458
+ register(K8S_SERVICES)
459
+ register(K8S_PODS)
460
+ register(K8S_TOP_PODS)
461
+ register(K8S_TOP_NODES)
462
+ register(K8S_HPA)
463
+ register(K8S_PVC)
464
+ register(K8S_CONTEXTS)
tools/monitoring.py ADDED
@@ -0,0 +1,162 @@
1
+ """Read-only monitoring/logging tools (Phase 8).
2
+
3
+ Query Prometheus (instant PromQL), Loki (instant LogQL) and Grafana health —
4
+ via the system curl, with the endpoint URL coming from ENVIRONMENT
5
+ CONFIGURATION only. This module is the one deliberate safety extension of
6
+ Phase 8, so its invariants are stricter than elsewhere:
7
+
8
+ - The endpoint can never come from the model. It is read from PROMETHEUS_URL,
9
+ LOKI_URL or GRAFANA_URL (set by the human running the agent); an unset
10
+ endpoint is an honest ToolError naming the variable. The model supplies
11
+ only the query text.
12
+ - The curl argv is fixed: `curl -fsS --max-time 10 --proto =https,http -H
13
+ Accept:application/json <url>`. GET only; `--proto` makes file:// and
14
+ other exotic protocols unreachable even in theory; there is no shell, so
15
+ the URL is one literal argv element.
16
+ - Query strings are percent-encoded with urllib.parse.quote(safe="") before
17
+ being appended, so `&`, spaces or shell metacharacters in a query become
18
+ part of the query, never of the URL structure or of a command.
19
+ - Endpoint URLs themselves are validated to start with http:// or https://
20
+ (a misconfigured PROMETHEUS_URL=file:///etc/passwd is refused, not run).
21
+
22
+ If no endpoint is configured, the tools fail with a clear message — nothing
23
+ is invented and no default host is ever contacted.
24
+ """
25
+
26
+ import os
27
+ from urllib.parse import quote
28
+
29
+ from tools.base import Tool, ToolError, read_command_output
30
+ from tools.registry import register
31
+
32
+ _TIMEOUT_S = 10
33
+ _MAX_QUERY_CHARS = 500
34
+
35
+
36
+ def _checked_query(args: dict, label: str) -> str:
37
+ query = args.get(label)
38
+ if not isinstance(query, str) or not query.strip():
39
+ raise ToolError(f"{label} must be a non-empty query string")
40
+ if len(query) > _MAX_QUERY_CHARS:
41
+ raise ToolError(f"{label} must be at most {_MAX_QUERY_CHARS} characters")
42
+ return query
43
+
44
+
45
+ def _endpoint(env_var: str) -> str:
46
+ base = os.environ.get(env_var, "").strip().rstrip("/")
47
+ if not base:
48
+ raise ToolError(
49
+ f"no monitoring endpoint configured: set the {env_var} "
50
+ "environment variable (e.g. http://prometheus:9090) and retry"
51
+ )
52
+ if not (base.startswith("http://") or base.startswith("https://")):
53
+ raise ToolError(
54
+ f"{env_var} must start with http:// or https:// (got {base!r})"
55
+ )
56
+ return base
57
+
58
+
59
+ def _curl(url: str) -> str:
60
+ return read_command_output(
61
+ (
62
+ "curl", "-fsS", "--max-time", "10", "--proto", "=https,http",
63
+ "-H", "Accept:application/json", url,
64
+ ),
65
+ timeout=_TIMEOUT_S,
66
+ )
67
+
68
+
69
+ def _prom_query(args: dict) -> str:
70
+ query = _checked_query(args, "query")
71
+ url = f"{_endpoint('PROMETHEUS_URL')}/api/v1/query?query={quote(query, safe='')}"
72
+ return _curl(url)
73
+
74
+
75
+ def _loki_query(args: dict) -> str:
76
+ query = _checked_query(args, "query")
77
+ try:
78
+ limit = int(args.get("limit", 100))
79
+ except (TypeError, ValueError):
80
+ raise ToolError("limit must be an integer between 1 and 1000")
81
+ if not 1 <= limit <= 1000:
82
+ raise ToolError("limit must be between 1 and 1000")
83
+ url = (
84
+ f"{_endpoint('LOKI_URL')}/loki/api/v1/query"
85
+ f"?query={quote(query, safe='')}&limit={limit}"
86
+ )
87
+ return _curl(url)
88
+
89
+
90
+ def _grafana_health(args: dict) -> str:
91
+ return _curl(f"{_endpoint('GRAFANA_URL')}/api/health")
92
+
93
+
94
+ PROM_QUERY = Tool(
95
+ name="prom_query",
96
+ description=(
97
+ "Run one instant PromQL query against the configured Prometheus "
98
+ "(PROMETHEUS_URL env) and return its JSON result: labels, values, "
99
+ "timestamps. Use for metrics evidence — error rate, latency, "
100
+ "saturation, replica counts. Query text only; the endpoint is fixed "
101
+ "by the operator's environment. Read-only."
102
+ ),
103
+ parameters={
104
+ "type": "object",
105
+ "properties": {
106
+ "query": {
107
+ "type": "string",
108
+ "description": "Instant PromQL query, e.g. "
109
+ "'sum(rate(http_requests_total{status=~\"5..\"}[5m]))'.",
110
+ }
111
+ },
112
+ "required": ["query"],
113
+ "additionalProperties": False,
114
+ },
115
+ executor=_prom_query,
116
+ )
117
+
118
+ LOKI_QUERY = Tool(
119
+ name="loki_query",
120
+ description=(
121
+ "Run one instant LogQL query against the configured Loki (LOKI_URL "
122
+ "env) and return its JSON result: log streams and lines. Use for log "
123
+ "evidence across many pods at once, e.g. "
124
+ "'{app=\"api\"} |= \"error\"'. Query text plus a bounded result "
125
+ "limit (1–1000); the endpoint is fixed by the environment. Read-only."
126
+ ),
127
+ parameters={
128
+ "type": "object",
129
+ "properties": {
130
+ "query": {
131
+ "type": "string",
132
+ "description": "Instant LogQL query (log or metric query).",
133
+ },
134
+ "limit": {
135
+ "type": "integer",
136
+ "minimum": 1,
137
+ "maximum": 1000,
138
+ "default": 100,
139
+ "description": "Max log entries to return (1–1000).",
140
+ },
141
+ },
142
+ "required": ["query"],
143
+ "additionalProperties": False,
144
+ },
145
+ executor=_loki_query,
146
+ )
147
+
148
+ GRAFANA_HEALTH = Tool(
149
+ name="grafana_health",
150
+ description=(
151
+ "Check the configured Grafana's health endpoint (GRAFANA_URL env, "
152
+ "/api/health): version and database status. Use to confirm whether "
153
+ "the observability stack itself is up — dashboards empty because "
154
+ "Grafana is down is its own incident. Read-only."
155
+ ),
156
+ parameters={"type": "object", "properties": {}, "additionalProperties": False},
157
+ executor=_grafana_health,
158
+ )
159
+
160
+ register(PROM_QUERY)
161
+ register(LOKI_QUERY)
162
+ register(GRAFANA_HEALTH)