observability-aiops 0.1.0__py3-none-any.whl
This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
- mcp_server/__init__.py +1 -0
- mcp_server/_shared.py +103 -0
- mcp_server/server.py +39 -0
- mcp_server/tools/__init__.py +1 -0
- mcp_server/tools/alerts.py +56 -0
- mcp_server/tools/analysis.py +59 -0
- mcp_server/tools/grafana.py +70 -0
- mcp_server/tools/metrics.py +72 -0
- mcp_server/tools/overview.py +22 -0
- mcp_server/tools/prometheus.py +31 -0
- mcp_server/tools/rules.py +32 -0
- mcp_server/tools/targets.py +44 -0
- mcp_server/tools/writes.py +203 -0
- observability_aiops/__init__.py +9 -0
- observability_aiops/cli/__init__.py +9 -0
- observability_aiops/cli/_common.py +78 -0
- observability_aiops/cli/_root.py +57 -0
- observability_aiops/cli/alert.py +49 -0
- observability_aiops/cli/doctor.py +21 -0
- observability_aiops/cli/init.py +135 -0
- observability_aiops/cli/overview.py +16 -0
- observability_aiops/cli/query.py +58 -0
- observability_aiops/cli/secret.py +107 -0
- observability_aiops/config.py +179 -0
- observability_aiops/connection.py +209 -0
- observability_aiops/doctor.py +103 -0
- observability_aiops/governance/__init__.py +40 -0
- observability_aiops/governance/audit.py +377 -0
- observability_aiops/governance/budget.py +225 -0
- observability_aiops/governance/decorators.py +474 -0
- observability_aiops/governance/paths.py +23 -0
- observability_aiops/governance/patterns.py +378 -0
- observability_aiops/governance/policy.py +411 -0
- observability_aiops/governance/sanitize.py +39 -0
- observability_aiops/governance/undo.py +218 -0
- observability_aiops/ops/__init__.py +1 -0
- observability_aiops/ops/_util.py +61 -0
- observability_aiops/ops/alerts.py +129 -0
- observability_aiops/ops/analysis.py +267 -0
- observability_aiops/ops/grafana.py +102 -0
- observability_aiops/ops/metrics.py +120 -0
- observability_aiops/ops/overview.py +76 -0
- observability_aiops/ops/prom_status.py +66 -0
- observability_aiops/ops/rules.py +80 -0
- observability_aiops/ops/targets.py +78 -0
- observability_aiops/ops/writes.py +170 -0
- observability_aiops/secretstore.py +302 -0
- observability_aiops-0.1.0.dist-info/METADATA +127 -0
- observability_aiops-0.1.0.dist-info/RECORD +52 -0
- observability_aiops-0.1.0.dist-info/WHEEL +4 -0
- observability_aiops-0.1.0.dist-info/entry_points.txt +3 -0
- observability_aiops-0.1.0.dist-info/licenses/LICENSE +21 -0
mcp_server/__init__.py
ADDED
|
@@ -0,0 +1 @@
|
|
|
1
|
+
"""MCP server package for observability-aiops."""
|
mcp_server/_shared.py
ADDED
|
@@ -0,0 +1,103 @@
|
|
|
1
|
+
"""Shared MCP server primitives: the FastMCP instance, connection helper,
|
|
2
|
+
error sanitisation, and the ``@tool_errors`` decorator.
|
|
3
|
+
|
|
4
|
+
Tool modules under ``mcp_server/tools/`` import ``mcp`` from here and register
|
|
5
|
+
their ``@mcp.tool()`` functions onto it. ``mcp_server/server.py`` then imports
|
|
6
|
+
those modules and runs the server.
|
|
7
|
+
|
|
8
|
+
Keep ``Optional[X]`` (never PEP 604 ``X | None``) in any FastMCP-reflected
|
|
9
|
+
tool signature — on older mcp/pydantic the union eval'd to ``types.UnionType``
|
|
10
|
+
crashes FastMCP's ``issubclass`` check.
|
|
11
|
+
"""
|
|
12
|
+
|
|
13
|
+
import functools
|
|
14
|
+
import logging
|
|
15
|
+
import os
|
|
16
|
+
from collections.abc import Callable
|
|
17
|
+
from pathlib import Path
|
|
18
|
+
from typing import Any, Optional
|
|
19
|
+
|
|
20
|
+
from mcp.server.fastmcp import FastMCP
|
|
21
|
+
|
|
22
|
+
from observability_aiops.config import load_config
|
|
23
|
+
from observability_aiops.connection import ConnectionManager, ObservabilityApiError
|
|
24
|
+
from observability_aiops.governance import sanitize
|
|
25
|
+
|
|
26
|
+
logger = logging.getLogger(__name__)
|
|
27
|
+
|
|
28
|
+
_DOCTOR_HINT = "Run 'observability-aiops doctor' to verify connectivity and credentials."
|
|
29
|
+
|
|
30
|
+
|
|
31
|
+
def _safe_error(exc: Exception, tool: str) -> str:
|
|
32
|
+
"""Return an agent-safe error string; log full detail server-side only."""
|
|
33
|
+
logger.error("Tool %s failed", tool, exc_info=True)
|
|
34
|
+
_passthrough = (
|
|
35
|
+
ValueError,
|
|
36
|
+
FileNotFoundError,
|
|
37
|
+
KeyError,
|
|
38
|
+
PermissionError,
|
|
39
|
+
TimeoutError,
|
|
40
|
+
ConnectionError,
|
|
41
|
+
ObservabilityApiError,
|
|
42
|
+
)
|
|
43
|
+
if isinstance(exc, _passthrough):
|
|
44
|
+
return sanitize(str(exc), 300)
|
|
45
|
+
return f"{type(exc).__name__}: operation failed."
|
|
46
|
+
|
|
47
|
+
|
|
48
|
+
def tool_errors(shape: str = "dict") -> Callable:
|
|
49
|
+
"""Wrap a tool body in the canonical try/except → ``_safe_error`` pattern.
|
|
50
|
+
|
|
51
|
+
Place this *between* ``@governed_tool`` and the function so the audit
|
|
52
|
+
decorator and FastMCP still see the original signature.
|
|
53
|
+
"""
|
|
54
|
+
|
|
55
|
+
def decorator(func: Callable) -> Callable:
|
|
56
|
+
name = func.__name__
|
|
57
|
+
|
|
58
|
+
@functools.wraps(func)
|
|
59
|
+
def wrapper(*args: Any, **kwargs: Any) -> Any:
|
|
60
|
+
try:
|
|
61
|
+
return func(*args, **kwargs)
|
|
62
|
+
except Exception as e: # noqa: BLE001 — sanitised below
|
|
63
|
+
msg = _safe_error(e, name)
|
|
64
|
+
if shape == "list":
|
|
65
|
+
return [{"error": msg, "hint": _DOCTOR_HINT}]
|
|
66
|
+
if shape == "str":
|
|
67
|
+
return f"Error: {msg} {_DOCTOR_HINT}"
|
|
68
|
+
return {"error": msg, "hint": _DOCTOR_HINT}
|
|
69
|
+
|
|
70
|
+
return wrapper
|
|
71
|
+
|
|
72
|
+
return decorator
|
|
73
|
+
|
|
74
|
+
|
|
75
|
+
mcp = FastMCP(
|
|
76
|
+
"observability-aiops",
|
|
77
|
+
instructions=(
|
|
78
|
+
"Self-hosted observability operations (preview) over Prometheus, "
|
|
79
|
+
"Alertmanager, and Grafana: PromQL instant/range queries, label + series "
|
|
80
|
+
"metadata; scrape-target health and dropped targets; recording/alerting "
|
|
81
|
+
"rules and their health; firing/pending alerts, Alertmanager alerts + "
|
|
82
|
+
"silences; Grafana dashboards, datasources, folders, and health; three "
|
|
83
|
+
"flagship analyses (firing-alert RCA, target-scrape-health, "
|
|
84
|
+
"alert-noise/flap); and governed writes — create/expire silence, create "
|
|
85
|
+
"annotation, update/delete dashboard, reload Prometheus config. Destructive "
|
|
86
|
+
"writes (delete dashboard) are risk=high with a dry_run preview and require "
|
|
87
|
+
"an approver. Reversible writes capture the real before-state and record an "
|
|
88
|
+
"undo. Every tool runs through the observability-aiops governance harness "
|
|
89
|
+
"(audit / budget / risk-tier / undo)."
|
|
90
|
+
),
|
|
91
|
+
)
|
|
92
|
+
|
|
93
|
+
_conn_mgr: Optional[ConnectionManager] = None
|
|
94
|
+
|
|
95
|
+
|
|
96
|
+
def _get_connection(target: Optional[str] = None) -> Any:
|
|
97
|
+
"""Return a Monitoring connection, lazily initialising the manager."""
|
|
98
|
+
global _conn_mgr # noqa: PLW0603
|
|
99
|
+
if _conn_mgr is None:
|
|
100
|
+
config_path_str = os.environ.get("OBSERVABILITY_AIOPS_CONFIG")
|
|
101
|
+
config_path = Path(config_path_str) if config_path_str else None
|
|
102
|
+
_conn_mgr = ConnectionManager(load_config(config_path))
|
|
103
|
+
return _conn_mgr.connect(target)
|
mcp_server/server.py
ADDED
|
@@ -0,0 +1,39 @@
|
|
|
1
|
+
"""MCP server wrapping observability-aiops operations (stdio transport).
|
|
2
|
+
|
|
3
|
+
Thin adapter layer: each ``@mcp.tool()`` function (in ``mcp_server/tools/``)
|
|
4
|
+
delegates to the ``observability_aiops`` ops package and is wrapped with the
|
|
5
|
+
observability-aiops ``@governed_tool`` harness (audit / budget / undo / risk-tier).
|
|
6
|
+
|
|
7
|
+
Standalone, self-governed self-hosted observability operations (preview) over
|
|
8
|
+
Prometheus, Alertmanager, and Grafana: PromQL, scrape-target + rule health,
|
|
9
|
+
alerts + silences, dashboards, flagship analyses, and governed writes.
|
|
10
|
+
|
|
11
|
+
Source: https://github.com/AIops-tools/Observability-AIops
|
|
12
|
+
License: MIT
|
|
13
|
+
"""
|
|
14
|
+
|
|
15
|
+
import logging
|
|
16
|
+
|
|
17
|
+
from mcp_server._shared import _safe_error, mcp, tool_errors
|
|
18
|
+
|
|
19
|
+
# Importing the tool modules registers every @mcp.tool() onto the shared
|
|
20
|
+
# `mcp` instance. Order does not matter; each module is self-contained.
|
|
21
|
+
from mcp_server.tools import ( # noqa: F401 — side effects
|
|
22
|
+
alerts,
|
|
23
|
+
analysis,
|
|
24
|
+
grafana,
|
|
25
|
+
metrics,
|
|
26
|
+
overview,
|
|
27
|
+
prometheus,
|
|
28
|
+
rules,
|
|
29
|
+
targets,
|
|
30
|
+
writes,
|
|
31
|
+
)
|
|
32
|
+
|
|
33
|
+
__all__ = ["mcp", "main", "_safe_error", "tool_errors"]
|
|
34
|
+
|
|
35
|
+
|
|
36
|
+
def main() -> None:
|
|
37
|
+
"""Run the MCP server over stdio."""
|
|
38
|
+
logging.basicConfig(level=logging.INFO)
|
|
39
|
+
mcp.run(transport="stdio")
|
|
@@ -0,0 +1 @@
|
|
|
1
|
+
"""MCP tool modules. Importing each registers its @mcp.tool() functions."""
|
|
@@ -0,0 +1,56 @@
|
|
|
1
|
+
"""Alert MCP tools: Prometheus rule alerts + Alertmanager alerts/silences (read-only)."""
|
|
2
|
+
|
|
3
|
+
from typing import Optional
|
|
4
|
+
|
|
5
|
+
from mcp_server._shared import _get_connection, mcp, tool_errors
|
|
6
|
+
from observability_aiops.governance import governed_tool
|
|
7
|
+
from observability_aiops.ops import alerts as ops
|
|
8
|
+
|
|
9
|
+
|
|
10
|
+
@mcp.tool()
|
|
11
|
+
@governed_tool(risk_level="low")
|
|
12
|
+
@tool_errors("dict")
|
|
13
|
+
def firing_alerts(target: Optional[str] = None) -> dict:
|
|
14
|
+
"""[READ] Currently firing Prometheus rule alerts, grouped by severity.
|
|
15
|
+
|
|
16
|
+
Args:
|
|
17
|
+
target: Prometheus target name from config; omit for the default.
|
|
18
|
+
"""
|
|
19
|
+
return ops.firing_alerts(_get_connection(target))
|
|
20
|
+
|
|
21
|
+
|
|
22
|
+
@mcp.tool()
|
|
23
|
+
@governed_tool(risk_level="low")
|
|
24
|
+
@tool_errors("dict")
|
|
25
|
+
def pending_alerts(target: Optional[str] = None) -> dict:
|
|
26
|
+
"""[READ] Pending (not-yet-firing) Prometheus rule alerts, by severity.
|
|
27
|
+
|
|
28
|
+
Args:
|
|
29
|
+
target: Prometheus target name from config; omit for the default.
|
|
30
|
+
"""
|
|
31
|
+
return ops.pending_alerts(_get_connection(target))
|
|
32
|
+
|
|
33
|
+
|
|
34
|
+
@mcp.tool()
|
|
35
|
+
@governed_tool(risk_level="low")
|
|
36
|
+
@tool_errors("dict")
|
|
37
|
+
def alertmanager_alerts(active_only: bool = True, target: Optional[str] = None) -> dict:
|
|
38
|
+
"""[READ] Alerts as Alertmanager sees them (post grouping/silence/inhibit).
|
|
39
|
+
|
|
40
|
+
Args:
|
|
41
|
+
active_only: If True, exclude silenced/inhibited alerts.
|
|
42
|
+
target: Prometheus target name from config (its Alertmanager); omit for default.
|
|
43
|
+
"""
|
|
44
|
+
return ops.alertmanager_alerts(_get_connection(target), active_only)
|
|
45
|
+
|
|
46
|
+
|
|
47
|
+
@mcp.tool()
|
|
48
|
+
@governed_tool(risk_level="low")
|
|
49
|
+
@tool_errors("dict")
|
|
50
|
+
def list_silences(target: Optional[str] = None) -> dict:
|
|
51
|
+
"""[READ] Alertmanager silences (active, pending, expired).
|
|
52
|
+
|
|
53
|
+
Args:
|
|
54
|
+
target: Prometheus target name from config (its Alertmanager); omit for default.
|
|
55
|
+
"""
|
|
56
|
+
return ops.list_silences(_get_connection(target))
|
|
@@ -0,0 +1,59 @@
|
|
|
1
|
+
"""Flagship analysis MCP tools — pull telemetry, then run the pure heuristics."""
|
|
2
|
+
|
|
3
|
+
from typing import Optional
|
|
4
|
+
|
|
5
|
+
from mcp_server._shared import _get_connection, mcp, tool_errors
|
|
6
|
+
from observability_aiops.governance import governed_tool
|
|
7
|
+
from observability_aiops.ops import alerts as alerts_ops
|
|
8
|
+
from observability_aiops.ops import analysis as ops
|
|
9
|
+
from observability_aiops.ops import rules as rules_ops
|
|
10
|
+
from observability_aiops.ops import targets as targets_ops
|
|
11
|
+
|
|
12
|
+
|
|
13
|
+
@mcp.tool()
|
|
14
|
+
@governed_tool(risk_level="low")
|
|
15
|
+
@tool_errors("dict")
|
|
16
|
+
def firing_alert_rca(target: Optional[str] = None) -> dict:
|
|
17
|
+
"""[READ][analysis] Root-cause firing alerts: join each to its rule expr → cause+action.
|
|
18
|
+
|
|
19
|
+
Pulls firing alerts + alerting rules, matches them, and maps each to a likely
|
|
20
|
+
cause and recommended action. Advisory heuristic — verify before acting.
|
|
21
|
+
|
|
22
|
+
Args:
|
|
23
|
+
target: Prometheus target name from config; omit for the default.
|
|
24
|
+
"""
|
|
25
|
+
conn = _get_connection(target)
|
|
26
|
+
firing = alerts_ops.pull_alerts(conn, state="firing")
|
|
27
|
+
rules = rules_ops.list_rules(conn, "alerting").get("rules", [])
|
|
28
|
+
return ops.firing_alert_rca(firing, rules)
|
|
29
|
+
|
|
30
|
+
|
|
31
|
+
@mcp.tool()
|
|
32
|
+
@governed_tool(risk_level="low")
|
|
33
|
+
@tool_errors("dict")
|
|
34
|
+
def target_scrape_health_analysis(target: Optional[str] = None) -> dict:
|
|
35
|
+
"""[READ][analysis] Rank down/erroring scrape targets and classify each cause.
|
|
36
|
+
|
|
37
|
+
Args:
|
|
38
|
+
target: Prometheus target name from config; omit for the default.
|
|
39
|
+
"""
|
|
40
|
+
conn = _get_connection(target)
|
|
41
|
+
active = targets_ops.list_targets(conn).get("targets", [])
|
|
42
|
+
return ops.target_scrape_health_analysis(active)
|
|
43
|
+
|
|
44
|
+
|
|
45
|
+
@mcp.tool()
|
|
46
|
+
@governed_tool(risk_level="low")
|
|
47
|
+
@tool_errors("dict")
|
|
48
|
+
def alert_noise_and_flap_analysis(
|
|
49
|
+
noise_threshold: int = ops.DEFAULT_NOISE_THRESHOLD, target: Optional[str] = None
|
|
50
|
+
) -> dict:
|
|
51
|
+
"""[READ][analysis] Find noisy/duplicate alerts → dedup/rollup recommendation.
|
|
52
|
+
|
|
53
|
+
Args:
|
|
54
|
+
noise_threshold: Instance count at/above which an alertname is "noisy".
|
|
55
|
+
target: Prometheus target name from config; omit for the default.
|
|
56
|
+
"""
|
|
57
|
+
conn = _get_connection(target)
|
|
58
|
+
stream = alerts_ops.pull_alerts(conn)
|
|
59
|
+
return ops.alert_noise_and_flap_analysis(stream, noise_threshold)
|
|
@@ -0,0 +1,70 @@
|
|
|
1
|
+
"""Grafana MCP tools (read-only): dashboards, datasources, folders."""
|
|
2
|
+
|
|
3
|
+
from typing import Optional
|
|
4
|
+
|
|
5
|
+
from mcp_server._shared import _get_connection, mcp, tool_errors
|
|
6
|
+
from observability_aiops.governance import governed_tool
|
|
7
|
+
from observability_aiops.ops import grafana as ops
|
|
8
|
+
|
|
9
|
+
|
|
10
|
+
@mcp.tool()
|
|
11
|
+
@governed_tool(risk_level="low")
|
|
12
|
+
@tool_errors("dict")
|
|
13
|
+
def list_dashboards(query: Optional[str] = None, target: Optional[str] = None) -> dict:
|
|
14
|
+
"""[READ] Grafana dashboards (optionally filtered by a title query).
|
|
15
|
+
|
|
16
|
+
Args:
|
|
17
|
+
query: Optional title substring to search for.
|
|
18
|
+
target: Grafana target name from config; omit for the default.
|
|
19
|
+
"""
|
|
20
|
+
return ops.list_dashboards(_get_connection(target), query)
|
|
21
|
+
|
|
22
|
+
|
|
23
|
+
@mcp.tool()
|
|
24
|
+
@governed_tool(risk_level="low")
|
|
25
|
+
@tool_errors("dict")
|
|
26
|
+
def get_dashboard(uid: str, target: Optional[str] = None) -> dict:
|
|
27
|
+
"""[READ] One dashboard's summary (title, version, panel + tag counts).
|
|
28
|
+
|
|
29
|
+
Args:
|
|
30
|
+
uid: Dashboard UID (from list_dashboards).
|
|
31
|
+
target: Grafana target name from config; omit for the default.
|
|
32
|
+
"""
|
|
33
|
+
return ops.get_dashboard(_get_connection(target), uid)
|
|
34
|
+
|
|
35
|
+
|
|
36
|
+
@mcp.tool()
|
|
37
|
+
@governed_tool(risk_level="low")
|
|
38
|
+
@tool_errors("dict")
|
|
39
|
+
def list_datasources(target: Optional[str] = None) -> dict:
|
|
40
|
+
"""[READ] Configured Grafana datasources (id, uid, name, type, default).
|
|
41
|
+
|
|
42
|
+
Args:
|
|
43
|
+
target: Grafana target name from config; omit for the default.
|
|
44
|
+
"""
|
|
45
|
+
return ops.list_datasources(_get_connection(target))
|
|
46
|
+
|
|
47
|
+
|
|
48
|
+
@mcp.tool()
|
|
49
|
+
@governed_tool(risk_level="low")
|
|
50
|
+
@tool_errors("dict")
|
|
51
|
+
def datasource_health(datasource_id: int, target: Optional[str] = None) -> dict:
|
|
52
|
+
"""[READ] Health of one Grafana datasource.
|
|
53
|
+
|
|
54
|
+
Args:
|
|
55
|
+
datasource_id: Numeric datasource id (from list_datasources).
|
|
56
|
+
target: Grafana target name from config; omit for the default.
|
|
57
|
+
"""
|
|
58
|
+
return ops.datasource_health(_get_connection(target), datasource_id)
|
|
59
|
+
|
|
60
|
+
|
|
61
|
+
@mcp.tool()
|
|
62
|
+
@governed_tool(risk_level="low")
|
|
63
|
+
@tool_errors("dict")
|
|
64
|
+
def list_folders(target: Optional[str] = None) -> dict:
|
|
65
|
+
"""[READ] Grafana folders.
|
|
66
|
+
|
|
67
|
+
Args:
|
|
68
|
+
target: Grafana target name from config; omit for the default.
|
|
69
|
+
"""
|
|
70
|
+
return ops.list_folders(_get_connection(target))
|
|
@@ -0,0 +1,72 @@
|
|
|
1
|
+
"""Prometheus metrics MCP tools (read-only PromQL + metadata)."""
|
|
2
|
+
|
|
3
|
+
from typing import Optional
|
|
4
|
+
|
|
5
|
+
from mcp_server._shared import _get_connection, mcp, tool_errors
|
|
6
|
+
from observability_aiops.governance import governed_tool
|
|
7
|
+
from observability_aiops.ops import metrics as ops
|
|
8
|
+
|
|
9
|
+
|
|
10
|
+
@mcp.tool()
|
|
11
|
+
@governed_tool(risk_level="low")
|
|
12
|
+
@tool_errors("dict")
|
|
13
|
+
def instant_query(query: str, time: Optional[str] = None, target: Optional[str] = None) -> dict:
|
|
14
|
+
"""[READ] Evaluate a PromQL expression at a single instant.
|
|
15
|
+
|
|
16
|
+
Args:
|
|
17
|
+
query: A PromQL expression (e.g. 'up' or 'rate(http_requests_total[5m])').
|
|
18
|
+
time: Optional RFC-3339 or unix timestamp for the evaluation instant.
|
|
19
|
+
target: Prometheus target name from config; omit for the default.
|
|
20
|
+
"""
|
|
21
|
+
return ops.instant_query(_get_connection(target), query, time)
|
|
22
|
+
|
|
23
|
+
|
|
24
|
+
@mcp.tool()
|
|
25
|
+
@governed_tool(risk_level="low")
|
|
26
|
+
@tool_errors("dict")
|
|
27
|
+
def range_query(
|
|
28
|
+
query: str, start: str, end: str, step: str = "60s", target: Optional[str] = None
|
|
29
|
+
) -> dict:
|
|
30
|
+
"""[READ] Evaluate a PromQL expression over a time range.
|
|
31
|
+
|
|
32
|
+
Args:
|
|
33
|
+
query: A PromQL expression.
|
|
34
|
+
start: Range start (RFC-3339 or unix timestamp).
|
|
35
|
+
end: Range end (RFC-3339 or unix timestamp).
|
|
36
|
+
step: Resolution step (e.g. '60s', '5m').
|
|
37
|
+
target: Prometheus target name from config; omit for the default.
|
|
38
|
+
"""
|
|
39
|
+
return ops.range_query(_get_connection(target), query, start, end, step)
|
|
40
|
+
|
|
41
|
+
|
|
42
|
+
@mcp.tool()
|
|
43
|
+
@governed_tool(risk_level="low")
|
|
44
|
+
@tool_errors("dict")
|
|
45
|
+
def label_values(
|
|
46
|
+
label: str = "__name__", match: Optional[str] = None, target: Optional[str] = None
|
|
47
|
+
) -> dict:
|
|
48
|
+
"""[READ] Distinct values of a label (default __name__ = all metric names).
|
|
49
|
+
|
|
50
|
+
Args:
|
|
51
|
+
label: Label name to enumerate (default __name__).
|
|
52
|
+
match: Optional PromQL selector to scope the values (e.g. '{job="api"}').
|
|
53
|
+
target: Prometheus target name from config; omit for the default.
|
|
54
|
+
"""
|
|
55
|
+
return ops.label_values(_get_connection(target), label, match)
|
|
56
|
+
|
|
57
|
+
|
|
58
|
+
@mcp.tool()
|
|
59
|
+
@governed_tool(risk_level="low")
|
|
60
|
+
@tool_errors("dict")
|
|
61
|
+
def series_metadata(
|
|
62
|
+
match: str, start: Optional[str] = None, end: Optional[str] = None, target: Optional[str] = None
|
|
63
|
+
) -> dict:
|
|
64
|
+
"""[READ] Series (label-set) metadata for a PromQL selector.
|
|
65
|
+
|
|
66
|
+
Args:
|
|
67
|
+
match: A PromQL series selector (e.g. 'up{job="node"}').
|
|
68
|
+
start: Optional range start (RFC-3339 or unix timestamp).
|
|
69
|
+
end: Optional range end (RFC-3339 or unix timestamp).
|
|
70
|
+
target: Prometheus target name from config; omit for the default.
|
|
71
|
+
"""
|
|
72
|
+
return ops.series_metadata(_get_connection(target), match, start, end)
|
|
@@ -0,0 +1,22 @@
|
|
|
1
|
+
"""One-shot observability overview MCP tool (read-only, platform-aware)."""
|
|
2
|
+
|
|
3
|
+
from typing import Optional
|
|
4
|
+
|
|
5
|
+
from mcp_server._shared import _get_connection, mcp, tool_errors
|
|
6
|
+
from observability_aiops.governance import governed_tool
|
|
7
|
+
from observability_aiops.ops import overview as ops
|
|
8
|
+
|
|
9
|
+
|
|
10
|
+
@mcp.tool()
|
|
11
|
+
@governed_tool(risk_level="low")
|
|
12
|
+
@tool_errors("dict")
|
|
13
|
+
def observability_overview(target: Optional[str] = None) -> dict:
|
|
14
|
+
"""[READ] Platform-aware health snapshot for the target.
|
|
15
|
+
|
|
16
|
+
Prometheus: firing-alert count + scrape up/down + rules erroring. Grafana:
|
|
17
|
+
dashboard / datasource / folder counts.
|
|
18
|
+
|
|
19
|
+
Args:
|
|
20
|
+
target: Target name from config; omit for the default.
|
|
21
|
+
"""
|
|
22
|
+
return ops.observability_overview(_get_connection(target))
|
|
@@ -0,0 +1,31 @@
|
|
|
1
|
+
"""Prometheus server-status MCP tools (read-only ``/api/v1/status``)."""
|
|
2
|
+
|
|
3
|
+
from typing import Optional
|
|
4
|
+
|
|
5
|
+
from mcp_server._shared import _get_connection, mcp, tool_errors
|
|
6
|
+
from observability_aiops.governance import governed_tool
|
|
7
|
+
from observability_aiops.ops import prom_status as ops
|
|
8
|
+
|
|
9
|
+
|
|
10
|
+
@mcp.tool()
|
|
11
|
+
@governed_tool(risk_level="low")
|
|
12
|
+
@tool_errors("dict")
|
|
13
|
+
def prometheus_config_status(target: Optional[str] = None) -> dict:
|
|
14
|
+
"""[READ] Running-config fingerprint + size (never the raw YAML/secrets).
|
|
15
|
+
|
|
16
|
+
Args:
|
|
17
|
+
target: Prometheus target name from config; omit for the default.
|
|
18
|
+
"""
|
|
19
|
+
return ops.config_status(_get_connection(target))
|
|
20
|
+
|
|
21
|
+
|
|
22
|
+
@mcp.tool()
|
|
23
|
+
@governed_tool(risk_level="low")
|
|
24
|
+
@tool_errors("dict")
|
|
25
|
+
def prometheus_tsdb_status(target: Optional[str] = None) -> dict:
|
|
26
|
+
"""[READ] TSDB head cardinality stats + the top metrics by series count.
|
|
27
|
+
|
|
28
|
+
Args:
|
|
29
|
+
target: Prometheus target name from config; omit for the default.
|
|
30
|
+
"""
|
|
31
|
+
return ops.tsdb_status(_get_connection(target))
|
|
@@ -0,0 +1,32 @@
|
|
|
1
|
+
"""Prometheus rules MCP tools (read-only)."""
|
|
2
|
+
|
|
3
|
+
from typing import Optional
|
|
4
|
+
|
|
5
|
+
from mcp_server._shared import _get_connection, mcp, tool_errors
|
|
6
|
+
from observability_aiops.governance import governed_tool
|
|
7
|
+
from observability_aiops.ops import rules as ops
|
|
8
|
+
|
|
9
|
+
|
|
10
|
+
@mcp.tool()
|
|
11
|
+
@governed_tool(risk_level="low")
|
|
12
|
+
@tool_errors("dict")
|
|
13
|
+
def list_rules(rule_type: Optional[str] = None, target: Optional[str] = None) -> dict:
|
|
14
|
+
"""[READ] All recording + alerting rules, optionally filtered by type.
|
|
15
|
+
|
|
16
|
+
Args:
|
|
17
|
+
rule_type: Filter by "alerting" or "recording"; omit for both.
|
|
18
|
+
target: Prometheus target name from config; omit for the default.
|
|
19
|
+
"""
|
|
20
|
+
return ops.list_rules(_get_connection(target), rule_type)
|
|
21
|
+
|
|
22
|
+
|
|
23
|
+
@mcp.tool()
|
|
24
|
+
@governed_tool(risk_level="low")
|
|
25
|
+
@tool_errors("dict")
|
|
26
|
+
def rule_health(target: Optional[str] = None) -> dict:
|
|
27
|
+
"""[READ] Rule-evaluation health summary + the list of erroring rules.
|
|
28
|
+
|
|
29
|
+
Args:
|
|
30
|
+
target: Prometheus target name from config; omit for the default.
|
|
31
|
+
"""
|
|
32
|
+
return ops.rule_health(_get_connection(target))
|
|
@@ -0,0 +1,44 @@
|
|
|
1
|
+
"""Prometheus scrape-target MCP tools (read-only)."""
|
|
2
|
+
|
|
3
|
+
from typing import Optional
|
|
4
|
+
|
|
5
|
+
from mcp_server._shared import _get_connection, mcp, tool_errors
|
|
6
|
+
from observability_aiops.governance import governed_tool
|
|
7
|
+
from observability_aiops.ops import targets as ops
|
|
8
|
+
|
|
9
|
+
|
|
10
|
+
@mcp.tool()
|
|
11
|
+
@governed_tool(risk_level="low")
|
|
12
|
+
@tool_errors("dict")
|
|
13
|
+
def list_targets(health: Optional[str] = None, target: Optional[str] = None) -> dict:
|
|
14
|
+
"""[READ] Active scrape targets, optionally filtered by health (up/down).
|
|
15
|
+
|
|
16
|
+
Args:
|
|
17
|
+
health: Filter by health state ("up" or "down"); omit for all.
|
|
18
|
+
target: Prometheus target name from config; omit for the default.
|
|
19
|
+
"""
|
|
20
|
+
return ops.list_targets(_get_connection(target), health)
|
|
21
|
+
|
|
22
|
+
|
|
23
|
+
@mcp.tool()
|
|
24
|
+
@governed_tool(risk_level="low")
|
|
25
|
+
@tool_errors("dict")
|
|
26
|
+
def target_scrape_health(target: Optional[str] = None) -> dict:
|
|
27
|
+
"""[READ] Up/down scrape-health summary plus the list of unhealthy targets.
|
|
28
|
+
|
|
29
|
+
Args:
|
|
30
|
+
target: Prometheus target name from config; omit for the default.
|
|
31
|
+
"""
|
|
32
|
+
return ops.target_scrape_health(_get_connection(target))
|
|
33
|
+
|
|
34
|
+
|
|
35
|
+
@mcp.tool()
|
|
36
|
+
@governed_tool(risk_level="low")
|
|
37
|
+
@tool_errors("dict")
|
|
38
|
+
def dropped_targets(target: Optional[str] = None) -> dict:
|
|
39
|
+
"""[READ] Targets discovered but dropped by relabeling.
|
|
40
|
+
|
|
41
|
+
Args:
|
|
42
|
+
target: Prometheus target name from config; omit for the default.
|
|
43
|
+
"""
|
|
44
|
+
return ops.dropped_targets(_get_connection(target))
|