pgtriage 0.1.0__py3-none-any.whl
This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
- pgtriage/__init__.py +9 -0
- pgtriage/__main__.py +3 -0
- pgtriage/analyzers/__init__.py +0 -0
- pgtriage/analyzers/config_rules.py +159 -0
- pgtriage/analyzers/explain.py +128 -0
- pgtriage/analyzers/patterns.py +182 -0
- pgtriage/collectors/__init__.py +0 -0
- pgtriage/collectors/config.py +63 -0
- pgtriage/collectors/index_health.py +84 -0
- pgtriage/collectors/slow_queries.py +76 -0
- pgtriage/collectors/table_health.py +72 -0
- pgtriage/connection.py +57 -0
- pgtriage/models.py +94 -0
- pgtriage/server.py +398 -0
- pgtriage-0.1.0.dist-info/METADATA +148 -0
- pgtriage-0.1.0.dist-info/RECORD +19 -0
- pgtriage-0.1.0.dist-info/WHEEL +4 -0
- pgtriage-0.1.0.dist-info/entry_points.txt +2 -0
- pgtriage-0.1.0.dist-info/licenses/LICENSE +21 -0
pgtriage/__init__.py
ADDED
pgtriage/__main__.py
ADDED
|
File without changes
|
|
@@ -0,0 +1,159 @@
|
|
|
1
|
+
"""Configuration recommendation rules."""
|
|
2
|
+
|
|
3
|
+
from pgtriage.models import Category, Finding, Severity
|
|
4
|
+
|
|
5
|
+
|
|
6
|
+
def _get_setting(settings: list[dict], name: str) -> str | None:
|
|
7
|
+
for s in settings:
|
|
8
|
+
if s["name"] == name:
|
|
9
|
+
return s["setting"]
|
|
10
|
+
return None
|
|
11
|
+
|
|
12
|
+
|
|
13
|
+
def _parse_memory_kb(value: str, unit: str | None) -> int:
|
|
14
|
+
"""Convert a pg_settings memory value to KB."""
|
|
15
|
+
num = int(value)
|
|
16
|
+
if unit == "8kB":
|
|
17
|
+
return num * 8
|
|
18
|
+
if unit == "kB":
|
|
19
|
+
return num
|
|
20
|
+
if unit == "MB":
|
|
21
|
+
return num * 1024
|
|
22
|
+
if unit == "GB":
|
|
23
|
+
return num * 1024 * 1024
|
|
24
|
+
return num
|
|
25
|
+
|
|
26
|
+
|
|
27
|
+
def _get_setting_with_unit(settings: list[dict], name: str) -> tuple[str | None, str | None]:
|
|
28
|
+
for s in settings:
|
|
29
|
+
if s["name"] == name:
|
|
30
|
+
return s["setting"], s.get("unit")
|
|
31
|
+
return None, None
|
|
32
|
+
|
|
33
|
+
|
|
34
|
+
def analyze_config(
|
|
35
|
+
settings: list[dict],
|
|
36
|
+
connection_stats: dict | None,
|
|
37
|
+
) -> list[Finding]:
|
|
38
|
+
findings = []
|
|
39
|
+
|
|
40
|
+
shared_buffers_val, shared_buffers_unit = _get_setting_with_unit(settings, "shared_buffers")
|
|
41
|
+
if shared_buffers_val and shared_buffers_unit:
|
|
42
|
+
shared_buffers_kb = _parse_memory_kb(shared_buffers_val, shared_buffers_unit)
|
|
43
|
+
if shared_buffers_kb < 128 * 1024:
|
|
44
|
+
findings.append(Finding(
|
|
45
|
+
severity=Severity.HIGH,
|
|
46
|
+
category=Category.CONFIG_ISSUE,
|
|
47
|
+
detail=(
|
|
48
|
+
f"shared_buffers is {shared_buffers_kb // 1024}MB. "
|
|
49
|
+
f"For production workloads, this should typically be 25% of available RAM "
|
|
50
|
+
f"(minimum 128MB for small instances)."
|
|
51
|
+
),
|
|
52
|
+
suggested_fix="ALTER SYSTEM SET shared_buffers = '256MB'; -- then restart PostgreSQL",
|
|
53
|
+
requires_downtime=True,
|
|
54
|
+
evidence={"shared_buffers_kb": shared_buffers_kb},
|
|
55
|
+
))
|
|
56
|
+
|
|
57
|
+
work_mem_val, work_mem_unit = _get_setting_with_unit(settings, "work_mem")
|
|
58
|
+
if work_mem_val and work_mem_unit:
|
|
59
|
+
work_mem_kb = _parse_memory_kb(work_mem_val, work_mem_unit)
|
|
60
|
+
if work_mem_kb <= 4 * 1024:
|
|
61
|
+
findings.append(Finding(
|
|
62
|
+
severity=Severity.LOW,
|
|
63
|
+
category=Category.CONFIG_ISSUE,
|
|
64
|
+
detail=(
|
|
65
|
+
f"work_mem is at default ({work_mem_kb // 1024}MB). "
|
|
66
|
+
f"Complex queries with sorts and hash joins may spill to disk. "
|
|
67
|
+
f"Consider increasing for workloads with complex queries."
|
|
68
|
+
),
|
|
69
|
+
suggested_fix="ALTER SYSTEM SET work_mem = '16MB'; -- then SELECT pg_reload_conf();",
|
|
70
|
+
evidence={"work_mem_kb": work_mem_kb},
|
|
71
|
+
))
|
|
72
|
+
|
|
73
|
+
autovacuum_sf = _get_setting(settings, "autovacuum_vacuum_scale_factor")
|
|
74
|
+
if autovacuum_sf:
|
|
75
|
+
sf_val = float(autovacuum_sf)
|
|
76
|
+
if sf_val > 0.1:
|
|
77
|
+
findings.append(Finding(
|
|
78
|
+
severity=Severity.MEDIUM,
|
|
79
|
+
category=Category.CONFIG_ISSUE,
|
|
80
|
+
detail=(
|
|
81
|
+
f"autovacuum_vacuum_scale_factor is {sf_val} (default 0.2). "
|
|
82
|
+
f"For large tables, this means autovacuum won't trigger until 20% of rows are dead. "
|
|
83
|
+
f"On a 10M row table, that's 2M dead rows before cleanup starts."
|
|
84
|
+
),
|
|
85
|
+
suggested_fix=(
|
|
86
|
+
"For high-churn tables, set per-table: "
|
|
87
|
+
"ALTER TABLE <table> SET (autovacuum_vacuum_scale_factor = 0.01);"
|
|
88
|
+
),
|
|
89
|
+
evidence={"autovacuum_vacuum_scale_factor": sf_val},
|
|
90
|
+
))
|
|
91
|
+
|
|
92
|
+
random_page_cost = _get_setting(settings, "random_page_cost")
|
|
93
|
+
if random_page_cost and float(random_page_cost) > 1.5:
|
|
94
|
+
findings.append(Finding(
|
|
95
|
+
severity=Severity.LOW,
|
|
96
|
+
category=Category.CONFIG_ISSUE,
|
|
97
|
+
detail=(
|
|
98
|
+
f"random_page_cost is {random_page_cost} (default 4.0). "
|
|
99
|
+
f"If your database is on SSD storage, a value of 1.1 better reflects "
|
|
100
|
+
f"actual random read performance and helps the planner choose index scans."
|
|
101
|
+
),
|
|
102
|
+
suggested_fix="ALTER SYSTEM SET random_page_cost = 1.1; -- then SELECT pg_reload_conf();",
|
|
103
|
+
evidence={"random_page_cost": float(random_page_cost)},
|
|
104
|
+
))
|
|
105
|
+
|
|
106
|
+
log_min_duration = _get_setting(settings, "log_min_duration_statement")
|
|
107
|
+
if log_min_duration and int(log_min_duration) < 0:
|
|
108
|
+
findings.append(Finding(
|
|
109
|
+
severity=Severity.INFO,
|
|
110
|
+
category=Category.CONFIG_ISSUE,
|
|
111
|
+
detail=(
|
|
112
|
+
"log_min_duration_statement is disabled (-1). "
|
|
113
|
+
"Enabling it helps identify slow queries in PostgreSQL logs."
|
|
114
|
+
),
|
|
115
|
+
suggested_fix=(
|
|
116
|
+
"ALTER SYSTEM SET log_min_duration_statement = 1000; "
|
|
117
|
+
"-- logs queries taking > 1 second"
|
|
118
|
+
),
|
|
119
|
+
evidence={"log_min_duration_statement": int(log_min_duration)},
|
|
120
|
+
))
|
|
121
|
+
|
|
122
|
+
if connection_stats:
|
|
123
|
+
total = connection_stats.get("total_connections", 0)
|
|
124
|
+
max_conn = connection_stats.get("max_connections", 100)
|
|
125
|
+
utilization = total / max(max_conn, 1) * 100
|
|
126
|
+
|
|
127
|
+
if utilization > 80:
|
|
128
|
+
findings.append(Finding(
|
|
129
|
+
severity=Severity.HIGH,
|
|
130
|
+
category=Category.CONNECTION_PRESSURE,
|
|
131
|
+
detail=(
|
|
132
|
+
f"Connection utilization at {utilization:.0f}% "
|
|
133
|
+
f"({total}/{max_conn}). "
|
|
134
|
+
f"Approaching max_connections limit."
|
|
135
|
+
),
|
|
136
|
+
suggested_fix=(
|
|
137
|
+
"Consider using a connection pooler (PgBouncer) or "
|
|
138
|
+
"increasing max_connections if RAM allows."
|
|
139
|
+
),
|
|
140
|
+
evidence={
|
|
141
|
+
"total_connections": total,
|
|
142
|
+
"max_connections": max_conn,
|
|
143
|
+
"utilization_pct": round(utilization, 1),
|
|
144
|
+
},
|
|
145
|
+
))
|
|
146
|
+
|
|
147
|
+
long_running = connection_stats.get("long_running_queries", 0)
|
|
148
|
+
if long_running > 0:
|
|
149
|
+
findings.append(Finding(
|
|
150
|
+
severity=Severity.HIGH,
|
|
151
|
+
category=Category.LONG_RUNNING_QUERY,
|
|
152
|
+
detail=(
|
|
153
|
+
f"{long_running} queries running for more than 30 seconds. "
|
|
154
|
+
f"Long-running queries hold locks and prevent autovacuum."
|
|
155
|
+
),
|
|
156
|
+
evidence={"long_running_queries": long_running},
|
|
157
|
+
))
|
|
158
|
+
|
|
159
|
+
return findings
|
|
@@ -0,0 +1,128 @@
|
|
|
1
|
+
"""EXPLAIN ANALYZE runner and execution plan analyzer."""
|
|
2
|
+
|
|
3
|
+
import re
|
|
4
|
+
|
|
5
|
+
from pgtriage.connection import ConnectionManager
|
|
6
|
+
from pgtriage.models import Category, Finding, Severity
|
|
7
|
+
|
|
8
|
+
SELECT_PATTERN = re.compile(r"^\s*SELECT\b", re.IGNORECASE)
|
|
9
|
+
STACKED_QUERY_PATTERN = re.compile(r";\s*\S")
|
|
10
|
+
|
|
11
|
+
|
|
12
|
+
async def run_explain_analyze(
|
|
13
|
+
db: ConnectionManager,
|
|
14
|
+
query: str,
|
|
15
|
+
) -> dict | None:
|
|
16
|
+
"""Run EXPLAIN (ANALYZE, BUFFERS, FORMAT JSON) on a SELECT query.
|
|
17
|
+
Returns the JSON plan or None if the query is not safe to run."""
|
|
18
|
+
if not SELECT_PATTERN.match(query):
|
|
19
|
+
return None
|
|
20
|
+
if STACKED_QUERY_PATTERN.search(query):
|
|
21
|
+
return None
|
|
22
|
+
|
|
23
|
+
clean_query = query.rstrip().rstrip(";")
|
|
24
|
+
explain_sql = f"EXPLAIN (ANALYZE, BUFFERS, FORMAT JSON) {clean_query}"
|
|
25
|
+
|
|
26
|
+
row = await db.fetch_one(explain_sql)
|
|
27
|
+
if row and "QUERY PLAN" in row:
|
|
28
|
+
return row["QUERY PLAN"]
|
|
29
|
+
return None
|
|
30
|
+
|
|
31
|
+
|
|
32
|
+
def detect_plan_issues(
|
|
33
|
+
plan_json: list[dict],
|
|
34
|
+
original_query: str | None = None,
|
|
35
|
+
) -> list[Finding]:
|
|
36
|
+
"""Analyze an EXPLAIN ANALYZE JSON plan for performance issues."""
|
|
37
|
+
if not plan_json:
|
|
38
|
+
return []
|
|
39
|
+
|
|
40
|
+
findings = []
|
|
41
|
+
plan = plan_json[0].get("Plan", {})
|
|
42
|
+
_walk_plan_node(plan, findings, original_query)
|
|
43
|
+
return findings
|
|
44
|
+
|
|
45
|
+
|
|
46
|
+
def _walk_plan_node(
|
|
47
|
+
node: dict,
|
|
48
|
+
findings: list[Finding],
|
|
49
|
+
original_query: str | None = None,
|
|
50
|
+
) -> None:
|
|
51
|
+
node_type = node.get("Node Type", "")
|
|
52
|
+
relation = node.get("Relation Name")
|
|
53
|
+
actual_rows = node.get("Actual Rows", 0)
|
|
54
|
+
plan_rows = node.get("Plan Rows", 0)
|
|
55
|
+
|
|
56
|
+
if node_type == "Seq Scan" and actual_rows > 100_000:
|
|
57
|
+
filter_text = node.get("Filter", "")
|
|
58
|
+
findings.append(Finding(
|
|
59
|
+
severity=Severity.HIGH if actual_rows > 1_000_000 else Severity.MEDIUM,
|
|
60
|
+
category=Category.SEQUENTIAL_SCAN,
|
|
61
|
+
table=relation,
|
|
62
|
+
query=original_query,
|
|
63
|
+
detail=(
|
|
64
|
+
f"Sequential scan on '{relation}' reading {actual_rows:,} rows. "
|
|
65
|
+
f"Filter: {filter_text or 'none'}. "
|
|
66
|
+
f"An index on the filtered columns would likely eliminate this scan."
|
|
67
|
+
),
|
|
68
|
+
estimated_impact=f"Scanning {actual_rows:,} rows instead of targeted index lookup",
|
|
69
|
+
suggested_fix=(
|
|
70
|
+
f"Identify the columns in the WHERE clause and create a targeted index: "
|
|
71
|
+
f"CREATE INDEX CONCURRENTLY ON {relation} (...);"
|
|
72
|
+
if relation else None
|
|
73
|
+
),
|
|
74
|
+
evidence={
|
|
75
|
+
"node_type": node_type,
|
|
76
|
+
"actual_rows": actual_rows,
|
|
77
|
+
"filter": filter_text,
|
|
78
|
+
"relation": relation,
|
|
79
|
+
},
|
|
80
|
+
))
|
|
81
|
+
|
|
82
|
+
if plan_rows > 0 and actual_rows > 0:
|
|
83
|
+
estimate_ratio = actual_rows / max(plan_rows, 1)
|
|
84
|
+
if estimate_ratio > 10 or estimate_ratio < 0.1:
|
|
85
|
+
findings.append(Finding(
|
|
86
|
+
severity=Severity.MEDIUM,
|
|
87
|
+
category=Category.STALE_STATS,
|
|
88
|
+
table=relation,
|
|
89
|
+
query=original_query,
|
|
90
|
+
detail=(
|
|
91
|
+
f"Row estimate is off by {estimate_ratio:.1f}x on '{relation or 'unknown'}'. "
|
|
92
|
+
f"Planned: {plan_rows:,}, actual: {actual_rows:,}. "
|
|
93
|
+
f"Table statistics may be stale, causing the planner to pick a bad strategy."
|
|
94
|
+
),
|
|
95
|
+
suggested_fix=f"ANALYZE {relation};" if relation else "Run ANALYZE on the relevant tables.",
|
|
96
|
+
evidence={
|
|
97
|
+
"planned_rows": plan_rows,
|
|
98
|
+
"actual_rows": actual_rows,
|
|
99
|
+
"estimate_ratio": round(estimate_ratio, 2),
|
|
100
|
+
"relation": relation,
|
|
101
|
+
},
|
|
102
|
+
))
|
|
103
|
+
|
|
104
|
+
if node_type == "Nested Loop" and actual_rows > 10_000:
|
|
105
|
+
inner = node.get("Plans", [{}])
|
|
106
|
+
inner_type = inner[-1].get("Node Type", "") if inner else ""
|
|
107
|
+
if inner_type == "Seq Scan":
|
|
108
|
+
inner_relation = inner[-1].get("Relation Name", "unknown")
|
|
109
|
+
findings.append(Finding(
|
|
110
|
+
severity=Severity.HIGH,
|
|
111
|
+
category=Category.MISSING_INDEX,
|
|
112
|
+
query=original_query,
|
|
113
|
+
detail=(
|
|
114
|
+
f"Nested loop join with sequential scan on '{inner_relation}' "
|
|
115
|
+
f"processing {actual_rows:,} rows. "
|
|
116
|
+
f"A hash join or index lookup would be faster."
|
|
117
|
+
),
|
|
118
|
+
estimated_impact="Nested loop + seq scan is the slowest join strategy",
|
|
119
|
+
evidence={
|
|
120
|
+
"outer_type": node_type,
|
|
121
|
+
"inner_type": inner_type,
|
|
122
|
+
"actual_rows": actual_rows,
|
|
123
|
+
"inner_relation": inner_relation,
|
|
124
|
+
},
|
|
125
|
+
))
|
|
126
|
+
|
|
127
|
+
for child in node.get("Plans", []):
|
|
128
|
+
_walk_plan_node(child, findings, original_query)
|
|
@@ -0,0 +1,182 @@
|
|
|
1
|
+
"""Deterministic pattern detection on PostgreSQL metrics."""
|
|
2
|
+
|
|
3
|
+
from datetime import datetime, timezone
|
|
4
|
+
|
|
5
|
+
from pgtriage.models import Category, Finding, Severity
|
|
6
|
+
|
|
7
|
+
DEAD_TUPLE_THRESHOLD_PCT = 10.0
|
|
8
|
+
LARGE_TABLE_ROWS = 100_000
|
|
9
|
+
SEQ_SCAN_RATIO_THRESHOLD = 80.0
|
|
10
|
+
VACUUM_STALE_HOURS = 24
|
|
11
|
+
TOAST_BLOAT_RATIO = 0.5
|
|
12
|
+
|
|
13
|
+
|
|
14
|
+
def analyze_dead_tuples(table_stats: list[dict]) -> list[Finding]:
|
|
15
|
+
findings = []
|
|
16
|
+
for t in table_stats:
|
|
17
|
+
dead_pct = float(t.get("dead_tuple_pct", 0))
|
|
18
|
+
dead_count = t.get("n_dead_tup", 0)
|
|
19
|
+
live_count = t.get("n_live_tup", 0)
|
|
20
|
+
|
|
21
|
+
if dead_pct <= DEAD_TUPLE_THRESHOLD_PCT or dead_count < 1000:
|
|
22
|
+
continue
|
|
23
|
+
|
|
24
|
+
if dead_pct > 30:
|
|
25
|
+
severity = Severity.CRITICAL
|
|
26
|
+
elif dead_pct > 20:
|
|
27
|
+
severity = Severity.HIGH
|
|
28
|
+
else:
|
|
29
|
+
severity = Severity.MEDIUM
|
|
30
|
+
|
|
31
|
+
findings.append(Finding(
|
|
32
|
+
severity=severity,
|
|
33
|
+
category=Category.DEAD_TUPLES,
|
|
34
|
+
table=t["table_name"],
|
|
35
|
+
detail=(
|
|
36
|
+
f"Table has {dead_count:,} dead tuples ({dead_pct}% of total rows). "
|
|
37
|
+
f"Live rows: {live_count:,}. "
|
|
38
|
+
f"Autovacuum may not be keeping up with the write volume."
|
|
39
|
+
),
|
|
40
|
+
estimated_impact="Table bloat increases query times and disk usage",
|
|
41
|
+
suggested_fix=(
|
|
42
|
+
f"VACUUM (VERBOSE) {t['table_name']}; "
|
|
43
|
+
f"-- or for severe cases: VACUUM FULL {t['table_name']}; "
|
|
44
|
+
f"(requires exclusive lock)"
|
|
45
|
+
),
|
|
46
|
+
safe_to_apply=True,
|
|
47
|
+
requires_downtime=False,
|
|
48
|
+
evidence={
|
|
49
|
+
"dead_tuples": dead_count,
|
|
50
|
+
"live_tuples": live_count,
|
|
51
|
+
"dead_tuple_pct": dead_pct,
|
|
52
|
+
"autovacuum_count": t.get("autovacuum_count", 0),
|
|
53
|
+
},
|
|
54
|
+
))
|
|
55
|
+
return findings
|
|
56
|
+
|
|
57
|
+
|
|
58
|
+
def analyze_sequential_scans(table_stats: list[dict]) -> list[Finding]:
|
|
59
|
+
findings = []
|
|
60
|
+
for t in table_stats:
|
|
61
|
+
live_rows = t.get("n_live_tup", 0)
|
|
62
|
+
seq_scan_pct = float(t.get("seq_scan_pct", 0))
|
|
63
|
+
seq_scans = t.get("seq_scan", 0)
|
|
64
|
+
idx_scans = t.get("idx_scan", 0)
|
|
65
|
+
|
|
66
|
+
if live_rows < LARGE_TABLE_ROWS:
|
|
67
|
+
continue
|
|
68
|
+
if seq_scan_pct < SEQ_SCAN_RATIO_THRESHOLD:
|
|
69
|
+
continue
|
|
70
|
+
if seq_scans < 100:
|
|
71
|
+
continue
|
|
72
|
+
|
|
73
|
+
if live_rows > 1_000_000:
|
|
74
|
+
severity = Severity.HIGH
|
|
75
|
+
else:
|
|
76
|
+
severity = Severity.MEDIUM
|
|
77
|
+
|
|
78
|
+
findings.append(Finding(
|
|
79
|
+
severity=severity,
|
|
80
|
+
category=Category.SEQUENTIAL_SCAN,
|
|
81
|
+
table=t["table_name"],
|
|
82
|
+
detail=(
|
|
83
|
+
f"Table has {live_rows:,} rows with {seq_scan_pct}% sequential scans "
|
|
84
|
+
f"({seq_scans:,} seq vs {idx_scans:,} idx). "
|
|
85
|
+
f"Likely missing an index on frequently queried columns."
|
|
86
|
+
),
|
|
87
|
+
estimated_impact="Sequential scans on large tables cause slow queries under load",
|
|
88
|
+
suggested_fix=(
|
|
89
|
+
f"Identify the most common WHERE clauses on {t['table_name']} "
|
|
90
|
+
f"and add targeted indexes with CREATE INDEX CONCURRENTLY."
|
|
91
|
+
),
|
|
92
|
+
evidence={
|
|
93
|
+
"live_rows": live_rows,
|
|
94
|
+
"seq_scans": seq_scans,
|
|
95
|
+
"idx_scans": idx_scans,
|
|
96
|
+
"seq_scan_pct": seq_scan_pct,
|
|
97
|
+
},
|
|
98
|
+
))
|
|
99
|
+
return findings
|
|
100
|
+
|
|
101
|
+
|
|
102
|
+
def analyze_vacuum_staleness(table_stats: list[dict]) -> list[Finding]:
|
|
103
|
+
findings = []
|
|
104
|
+
now = datetime.now(timezone.utc)
|
|
105
|
+
|
|
106
|
+
for t in table_stats:
|
|
107
|
+
dead_count = t.get("n_dead_tup", 0)
|
|
108
|
+
if dead_count < 1000:
|
|
109
|
+
continue
|
|
110
|
+
|
|
111
|
+
last_vacuum = t.get("last_autovacuum") or t.get("last_vacuum")
|
|
112
|
+
if last_vacuum is None:
|
|
113
|
+
findings.append(Finding(
|
|
114
|
+
severity=Severity.MEDIUM,
|
|
115
|
+
category=Category.AUTOVACUUM_LAG,
|
|
116
|
+
table=t["table_name"],
|
|
117
|
+
detail=(
|
|
118
|
+
f"Table has never been vacuumed but has {dead_count:,} dead tuples. "
|
|
119
|
+
f"Autovacuum may not be configured or may not have triggered yet."
|
|
120
|
+
),
|
|
121
|
+
suggested_fix=f"VACUUM ANALYZE {t['table_name']};",
|
|
122
|
+
evidence={"dead_tuples": dead_count, "last_vacuum": None},
|
|
123
|
+
))
|
|
124
|
+
continue
|
|
125
|
+
|
|
126
|
+
if last_vacuum.tzinfo is None:
|
|
127
|
+
last_vacuum = last_vacuum.replace(tzinfo=timezone.utc)
|
|
128
|
+
|
|
129
|
+
hours_since = (now - last_vacuum).total_seconds() / 3600
|
|
130
|
+
if hours_since > VACUUM_STALE_HOURS and dead_count > 10_000:
|
|
131
|
+
findings.append(Finding(
|
|
132
|
+
severity=Severity.MEDIUM,
|
|
133
|
+
category=Category.AUTOVACUUM_LAG,
|
|
134
|
+
table=t["table_name"],
|
|
135
|
+
detail=(
|
|
136
|
+
f"Table has {dead_count:,} dead tuples and hasn't been vacuumed "
|
|
137
|
+
f"in {hours_since:.0f} hours. Last vacuum: {last_vacuum}."
|
|
138
|
+
),
|
|
139
|
+
suggested_fix=f"VACUUM ANALYZE {t['table_name']};",
|
|
140
|
+
evidence={
|
|
141
|
+
"dead_tuples": dead_count,
|
|
142
|
+
"hours_since_vacuum": round(hours_since, 1),
|
|
143
|
+
"last_vacuum": str(last_vacuum),
|
|
144
|
+
},
|
|
145
|
+
))
|
|
146
|
+
return findings
|
|
147
|
+
|
|
148
|
+
|
|
149
|
+
def analyze_table_bloat(table_sizes: list[dict]) -> list[Finding]:
|
|
150
|
+
findings = []
|
|
151
|
+
for t in table_sizes:
|
|
152
|
+
table_bytes = t.get("table_size_bytes", 0)
|
|
153
|
+
toast_bytes = t.get("toast_size_bytes", 0)
|
|
154
|
+
|
|
155
|
+
if table_bytes < 10_000_000:
|
|
156
|
+
continue
|
|
157
|
+
if toast_bytes <= 0:
|
|
158
|
+
continue
|
|
159
|
+
|
|
160
|
+
toast_ratio = toast_bytes / max(table_bytes, 1)
|
|
161
|
+
if toast_ratio < TOAST_BLOAT_RATIO:
|
|
162
|
+
continue
|
|
163
|
+
|
|
164
|
+
findings.append(Finding(
|
|
165
|
+
severity=Severity.MEDIUM,
|
|
166
|
+
category=Category.TOAST_BLOAT,
|
|
167
|
+
table=t["table_name"],
|
|
168
|
+
detail=(
|
|
169
|
+
f"TOAST storage ({t.get('toast_size', 'N/A')}) is "
|
|
170
|
+
f"{toast_ratio:.1f}x the table size ({t.get('table_size', 'N/A')}). "
|
|
171
|
+
f"Large JSONB or TEXT columns may be causing bloat. "
|
|
172
|
+
f"TOAST tables have separate autovacuum tracking and can lag behind."
|
|
173
|
+
),
|
|
174
|
+
estimated_impact="TOAST bloat increases disk usage and slows full table operations",
|
|
175
|
+
evidence={
|
|
176
|
+
"table_size": t.get("table_size"),
|
|
177
|
+
"toast_size": t.get("toast_size"),
|
|
178
|
+
"total_size": t.get("total_size"),
|
|
179
|
+
"toast_ratio": round(toast_ratio, 2),
|
|
180
|
+
},
|
|
181
|
+
))
|
|
182
|
+
return findings
|
|
File without changes
|
|
@@ -0,0 +1,63 @@
|
|
|
1
|
+
"""Collect PostgreSQL configuration and connection stats."""
|
|
2
|
+
|
|
3
|
+
from pgtriage.connection import ConnectionManager
|
|
4
|
+
|
|
5
|
+
CONFIG_SETTINGS_QUERY = """
|
|
6
|
+
SELECT name, setting, unit, short_desc, context, boot_val, reset_val
|
|
7
|
+
FROM pg_settings
|
|
8
|
+
WHERE name IN (
|
|
9
|
+
'max_connections',
|
|
10
|
+
'shared_buffers',
|
|
11
|
+
'effective_cache_size',
|
|
12
|
+
'work_mem',
|
|
13
|
+
'maintenance_work_mem',
|
|
14
|
+
'random_page_cost',
|
|
15
|
+
'seq_page_cost',
|
|
16
|
+
'default_statistics_target',
|
|
17
|
+
'checkpoint_completion_target',
|
|
18
|
+
'wal_buffers',
|
|
19
|
+
'min_wal_size',
|
|
20
|
+
'max_wal_size',
|
|
21
|
+
'autovacuum',
|
|
22
|
+
'autovacuum_max_workers',
|
|
23
|
+
'autovacuum_vacuum_threshold',
|
|
24
|
+
'autovacuum_vacuum_scale_factor',
|
|
25
|
+
'autovacuum_analyze_threshold',
|
|
26
|
+
'autovacuum_analyze_scale_factor',
|
|
27
|
+
'autovacuum_vacuum_cost_delay',
|
|
28
|
+
'autovacuum_vacuum_cost_limit',
|
|
29
|
+
'log_min_duration_statement',
|
|
30
|
+
'track_activity_query_size'
|
|
31
|
+
)
|
|
32
|
+
ORDER BY name
|
|
33
|
+
"""
|
|
34
|
+
|
|
35
|
+
CONNECTION_STATS_QUERY = """
|
|
36
|
+
SELECT
|
|
37
|
+
(SELECT count(*) FROM pg_stat_activity) AS total_connections,
|
|
38
|
+
(SELECT setting::int FROM pg_settings WHERE name = 'max_connections') AS max_connections,
|
|
39
|
+
(SELECT count(*) FROM pg_stat_activity WHERE state = 'active') AS active_queries,
|
|
40
|
+
(SELECT count(*) FROM pg_stat_activity WHERE state = 'idle') AS idle_connections,
|
|
41
|
+
(SELECT count(*) FROM pg_stat_activity
|
|
42
|
+
WHERE state = 'active'
|
|
43
|
+
AND now() - query_start > interval '30 seconds') AS long_running_queries
|
|
44
|
+
"""
|
|
45
|
+
|
|
46
|
+
DATABASE_INFO_QUERY = """
|
|
47
|
+
SELECT
|
|
48
|
+
pg_size_pretty(pg_database_size(current_database())) AS database_size,
|
|
49
|
+
current_database() AS database_name,
|
|
50
|
+
(SELECT setting FROM pg_settings WHERE name = 'server_version') AS pg_version
|
|
51
|
+
"""
|
|
52
|
+
|
|
53
|
+
|
|
54
|
+
async def collect_config_settings(db: ConnectionManager) -> list[dict]:
|
|
55
|
+
return await db.fetch_all(CONFIG_SETTINGS_QUERY)
|
|
56
|
+
|
|
57
|
+
|
|
58
|
+
async def collect_connection_stats(db: ConnectionManager) -> dict | None:
|
|
59
|
+
return await db.fetch_one(CONNECTION_STATS_QUERY)
|
|
60
|
+
|
|
61
|
+
|
|
62
|
+
async def collect_database_info(db: ConnectionManager) -> dict | None:
|
|
63
|
+
return await db.fetch_one(DATABASE_INFO_QUERY)
|
|
@@ -0,0 +1,84 @@
|
|
|
1
|
+
"""Collect index health metrics from PostgreSQL system views."""
|
|
2
|
+
|
|
3
|
+
from pgtriage.connection import ConnectionManager
|
|
4
|
+
|
|
5
|
+
UNUSED_INDEXES_QUERY = """
|
|
6
|
+
SELECT
|
|
7
|
+
s.schemaname,
|
|
8
|
+
s.relname AS table_name,
|
|
9
|
+
s.indexrelname AS index_name,
|
|
10
|
+
s.idx_scan AS scans_since_reset,
|
|
11
|
+
pg_size_pretty(pg_relation_size(s.indexrelid)) AS index_size,
|
|
12
|
+
pg_relation_size(s.indexrelid) AS index_size_bytes,
|
|
13
|
+
pg_get_indexdef(i.indexrelid) AS index_definition,
|
|
14
|
+
i.indisunique AS is_unique,
|
|
15
|
+
i.indisprimary AS is_primary
|
|
16
|
+
FROM pg_stat_user_indexes s
|
|
17
|
+
JOIN pg_index i ON s.indexrelid = i.indexrelid
|
|
18
|
+
WHERE s.idx_scan = 0
|
|
19
|
+
AND NOT i.indisunique
|
|
20
|
+
AND NOT i.indisprimary
|
|
21
|
+
AND s.schemaname = %s
|
|
22
|
+
ORDER BY pg_relation_size(s.indexrelid) DESC
|
|
23
|
+
"""
|
|
24
|
+
|
|
25
|
+
DUPLICATE_INDEXES_QUERY = """
|
|
26
|
+
SELECT
|
|
27
|
+
a.indrelid::regclass AS table_name,
|
|
28
|
+
a.indexrelid::regclass AS index_1,
|
|
29
|
+
b.indexrelid::regclass AS index_2,
|
|
30
|
+
pg_get_indexdef(a.indexrelid) AS index_1_def,
|
|
31
|
+
pg_get_indexdef(b.indexrelid) AS index_2_def,
|
|
32
|
+
pg_size_pretty(pg_relation_size(a.indexrelid)) AS index_1_size,
|
|
33
|
+
pg_size_pretty(pg_relation_size(b.indexrelid)) AS index_2_size,
|
|
34
|
+
pg_relation_size(a.indexrelid) AS index_1_size_bytes,
|
|
35
|
+
pg_relation_size(b.indexrelid) AS index_2_size_bytes
|
|
36
|
+
FROM pg_index a
|
|
37
|
+
JOIN pg_index b ON a.indrelid = b.indrelid
|
|
38
|
+
AND a.indexrelid < b.indexrelid
|
|
39
|
+
AND a.indkey::text = b.indkey::text
|
|
40
|
+
JOIN pg_namespace n ON n.oid = (
|
|
41
|
+
SELECT relnamespace FROM pg_class WHERE oid = a.indrelid
|
|
42
|
+
)
|
|
43
|
+
WHERE n.nspname = %s
|
|
44
|
+
ORDER BY pg_relation_size(a.indexrelid) + pg_relation_size(b.indexrelid) DESC
|
|
45
|
+
"""
|
|
46
|
+
|
|
47
|
+
TABLES_NEEDING_INDEXES_QUERY = """
|
|
48
|
+
SELECT
|
|
49
|
+
schemaname,
|
|
50
|
+
relname AS table_name,
|
|
51
|
+
seq_scan,
|
|
52
|
+
idx_scan,
|
|
53
|
+
n_live_tup,
|
|
54
|
+
seq_tup_read,
|
|
55
|
+
CASE WHEN seq_scan > 0
|
|
56
|
+
THEN seq_tup_read / seq_scan
|
|
57
|
+
ELSE 0 END AS avg_rows_per_seq_scan
|
|
58
|
+
FROM pg_stat_user_tables
|
|
59
|
+
WHERE n_live_tup > 100000
|
|
60
|
+
AND seq_scan > idx_scan
|
|
61
|
+
AND seq_scan > 100
|
|
62
|
+
ORDER BY seq_tup_read DESC
|
|
63
|
+
LIMIT 20
|
|
64
|
+
"""
|
|
65
|
+
|
|
66
|
+
|
|
67
|
+
async def collect_unused_indexes(
|
|
68
|
+
db: ConnectionManager,
|
|
69
|
+
schema_name: str = "public",
|
|
70
|
+
) -> list[dict]:
|
|
71
|
+
return await db.fetch_all(UNUSED_INDEXES_QUERY, (schema_name,))
|
|
72
|
+
|
|
73
|
+
|
|
74
|
+
async def collect_duplicate_indexes(
|
|
75
|
+
db: ConnectionManager,
|
|
76
|
+
schema_name: str = "public",
|
|
77
|
+
) -> list[dict]:
|
|
78
|
+
return await db.fetch_all(DUPLICATE_INDEXES_QUERY, (schema_name,))
|
|
79
|
+
|
|
80
|
+
|
|
81
|
+
async def collect_tables_needing_indexes(
|
|
82
|
+
db: ConnectionManager,
|
|
83
|
+
) -> list[dict]:
|
|
84
|
+
return await db.fetch_all(TABLES_NEEDING_INDEXES_QUERY)
|