pytest-querycount 0.3.0__py3-none-any.whl
This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
- pytest_querycount/__init__.py +31 -0
- pytest_querycount/backends/__init__.py +59 -0
- pytest_querycount/backends/sqlalchemy.py +215 -0
- pytest_querycount/checks.py +166 -0
- pytest_querycount/errors.py +39 -0
- pytest_querycount/explain.py +175 -0
- pytest_querycount/normalize.py +97 -0
- pytest_querycount/plugin.py +284 -0
- pytest_querycount/py.typed +0 -0
- pytest_querycount/recorder.py +214 -0
- pytest_querycount/records.py +66 -0
- pytest_querycount/report.py +64 -0
- pytest_querycount-0.3.0.dist-info/METADATA +336 -0
- pytest_querycount-0.3.0.dist-info/RECORD +17 -0
- pytest_querycount-0.3.0.dist-info/WHEEL +4 -0
- pytest_querycount-0.3.0.dist-info/entry_points.txt +2 -0
- pytest_querycount-0.3.0.dist-info/licenses/LICENSE +21 -0
|
@@ -0,0 +1,97 @@
|
|
|
1
|
+
"""Reduce a SQL statement to a *fingerprint*.
|
|
2
|
+
|
|
3
|
+
Two statements share a fingerprint when they differ only in their literal
|
|
4
|
+
values. That is what lets us say "this same query ran 47 times" instead of
|
|
5
|
+
"47 queries ran", which is the difference between a number and a diagnosis.
|
|
6
|
+
|
|
7
|
+
>>> fingerprint("SELECT * FROM users WHERE id = 42")
|
|
8
|
+
'select * from users where id = ?'
|
|
9
|
+
>>> fingerprint("SELECT * FROM users WHERE id = 43")
|
|
10
|
+
'select * from users where id = ?'
|
|
11
|
+
"""
|
|
12
|
+
|
|
13
|
+
from __future__ import annotations
|
|
14
|
+
|
|
15
|
+
import re
|
|
16
|
+
from functools import lru_cache
|
|
17
|
+
|
|
18
|
+
_PLACEHOLDER = "?"
|
|
19
|
+
|
|
20
|
+
# Comments go first: they can contain anything, including quotes that would
|
|
21
|
+
# otherwise unbalance the string matching below.
|
|
22
|
+
_BLOCK_COMMENT = re.compile(r"/\*.*?\*/", re.DOTALL)
|
|
23
|
+
_LINE_COMMENT = re.compile(r"--[^\n]*")
|
|
24
|
+
|
|
25
|
+
# Postgres dollar quoting ($$body$$ or $tag$body$tag$) must be handled before
|
|
26
|
+
# numeric placeholders, or "$1" inside a body would be mangled.
|
|
27
|
+
_DOLLAR_QUOTED = re.compile(r"\$(\w*)\$.*?\$\1\$", re.DOTALL)
|
|
28
|
+
|
|
29
|
+
# Single-quoted strings, with '' as the escape for an embedded quote.
|
|
30
|
+
_STRING = re.compile(r"'(?:[^']|'')*'")
|
|
31
|
+
|
|
32
|
+
# Bind parameters in every style the DBAPIs use. Named styles come before the
|
|
33
|
+
# bare ones so ":name" is not left as a stray colon.
|
|
34
|
+
_PARAM_PYFORMAT = re.compile(r"%\(\w+\)s")
|
|
35
|
+
_PARAM_NAMED = re.compile(r"(?<![:\w]):\w+") # ":name" but never "::cast"
|
|
36
|
+
_PARAM_NUMERIC = re.compile(r"\$\d+")
|
|
37
|
+
_PARAM_FORMAT = re.compile(r"%s")
|
|
38
|
+
|
|
39
|
+
# Numbers, once strings are gone so we never touch digits inside a literal.
|
|
40
|
+
# The word boundary keeps identifiers like "table1" intact.
|
|
41
|
+
_NUMBER = re.compile(r"\b\d+\.?\d*(?:[eE][+-]?\d+)?\b")
|
|
42
|
+
|
|
43
|
+
# Collapse variable-length lists so an IN clause of 3 ids fingerprints the same
|
|
44
|
+
# as one of 300 -- otherwise every batch size looks like a different query.
|
|
45
|
+
_IN_LIST = re.compile(r"\bin\s*\(\s*\?(?:\s*,\s*\?)*\s*\)")
|
|
46
|
+
_VALUES_TUPLE = re.compile(r"\(\s*\?(?:\s*,\s*\?)*\s*\)")
|
|
47
|
+
_VALUES_CLAUSE = re.compile(r"\bvalues\s*((?:\(\s*\?(?:\s*,\s*\?)*\s*\)\s*,?\s*)+)")
|
|
48
|
+
|
|
49
|
+
_WHITESPACE = re.compile(r"\s+")
|
|
50
|
+
|
|
51
|
+
|
|
52
|
+
def _collapse_values(match: re.Match[str]) -> str:
|
|
53
|
+
"""Rewrite a multi-row VALUES clause as a single tuple."""
|
|
54
|
+
tuples = _VALUES_TUPLE.findall(match.group(1))
|
|
55
|
+
if len(tuples) <= 1:
|
|
56
|
+
return match.group(0)
|
|
57
|
+
first = _VALUES_TUPLE.search(match.group(1))
|
|
58
|
+
assert first is not None
|
|
59
|
+
return f"values {first.group(0)}"
|
|
60
|
+
|
|
61
|
+
|
|
62
|
+
@lru_cache(maxsize=2048)
|
|
63
|
+
def fingerprint(statement: str) -> str:
|
|
64
|
+
"""Return a stable, literal-free form of ``statement``.
|
|
65
|
+
|
|
66
|
+
Cached, because the whole point is that the same statements recur and this
|
|
67
|
+
runs on every query the suite emits.
|
|
68
|
+
"""
|
|
69
|
+
sql = _BLOCK_COMMENT.sub(" ", statement)
|
|
70
|
+
sql = _LINE_COMMENT.sub(" ", sql)
|
|
71
|
+
sql = _DOLLAR_QUOTED.sub(_PLACEHOLDER, sql)
|
|
72
|
+
sql = _STRING.sub(_PLACEHOLDER, sql)
|
|
73
|
+
sql = _PARAM_PYFORMAT.sub(_PLACEHOLDER, sql)
|
|
74
|
+
sql = _PARAM_NAMED.sub(_PLACEHOLDER, sql)
|
|
75
|
+
sql = _PARAM_NUMERIC.sub(_PLACEHOLDER, sql)
|
|
76
|
+
sql = _PARAM_FORMAT.sub(_PLACEHOLDER, sql)
|
|
77
|
+
sql = _NUMBER.sub(_PLACEHOLDER, sql)
|
|
78
|
+
|
|
79
|
+
sql = sql.lower()
|
|
80
|
+
sql = _IN_LIST.sub(f"in ({_PLACEHOLDER})", sql)
|
|
81
|
+
sql = _VALUES_CLAUSE.sub(_collapse_values, sql)
|
|
82
|
+
|
|
83
|
+
sql = _WHITESPACE.sub(" ", sql).strip()
|
|
84
|
+
return sql.rstrip(";").strip()
|
|
85
|
+
|
|
86
|
+
|
|
87
|
+
def statement_kind(statement: str) -> str:
|
|
88
|
+
"""The leading keyword of ``statement``, lowercased.
|
|
89
|
+
|
|
90
|
+
``"select"``, ``"insert"``, ``"savepoint"``, ... or ``"other"`` when the
|
|
91
|
+
statement does not start with a word (an empty string, say).
|
|
92
|
+
"""
|
|
93
|
+
for token in _WHITESPACE.split(_LINE_COMMENT.sub(" ", statement).strip()):
|
|
94
|
+
cleaned = token.strip("(").lower()
|
|
95
|
+
if cleaned.isalpha():
|
|
96
|
+
return cleaned
|
|
97
|
+
return "other"
|
|
@@ -0,0 +1,284 @@
|
|
|
1
|
+
"""pytest integration: options, markers, the fixture, and the summary."""
|
|
2
|
+
|
|
3
|
+
from __future__ import annotations
|
|
4
|
+
|
|
5
|
+
from collections.abc import Generator, Sequence
|
|
6
|
+
from typing import Any
|
|
7
|
+
|
|
8
|
+
import pytest
|
|
9
|
+
|
|
10
|
+
from pytest_querycount import backends, checks, report
|
|
11
|
+
from pytest_querycount.checks import DEFAULT_DUPLICATE_THRESHOLD, DEFAULT_KINDS, Budget
|
|
12
|
+
from pytest_querycount.recorder import Recorder
|
|
13
|
+
|
|
14
|
+
RECORDER_KEY = pytest.StashKey[Recorder]()
|
|
15
|
+
_STATS: dict[str, report.TestStats] = {}
|
|
16
|
+
|
|
17
|
+
|
|
18
|
+
# -- configuration ---------------------------------------------------------
|
|
19
|
+
|
|
20
|
+
|
|
21
|
+
def pytest_addoption(parser: pytest.Parser) -> None:
|
|
22
|
+
group = parser.getgroup("querycount", "SQL query budgets")
|
|
23
|
+
group.addoption(
|
|
24
|
+
"--querycount-report",
|
|
25
|
+
action="store_true",
|
|
26
|
+
default=False,
|
|
27
|
+
help="Print a table of the tests that ran the most SQL queries.",
|
|
28
|
+
)
|
|
29
|
+
group.addoption(
|
|
30
|
+
"--querycount-top",
|
|
31
|
+
type=int,
|
|
32
|
+
default=10,
|
|
33
|
+
metavar="N",
|
|
34
|
+
help="How many tests the report lists (default: 10).",
|
|
35
|
+
)
|
|
36
|
+
group.addoption(
|
|
37
|
+
"--querycount-max",
|
|
38
|
+
type=int,
|
|
39
|
+
default=None,
|
|
40
|
+
metavar="N",
|
|
41
|
+
help="Apply a query budget of N to every test that has no explicit one.",
|
|
42
|
+
)
|
|
43
|
+
group.addoption(
|
|
44
|
+
"--querycount-no-seq-scan",
|
|
45
|
+
action="store_true",
|
|
46
|
+
default=False,
|
|
47
|
+
help=(
|
|
48
|
+
"Apply the missing-index check to every test. Costs one EXPLAIN per "
|
|
49
|
+
"SELECT, so prefer the marker once you know where to look."
|
|
50
|
+
),
|
|
51
|
+
)
|
|
52
|
+
parser.addini(
|
|
53
|
+
"querycount_max",
|
|
54
|
+
help="Default query budget for tests without an explicit one.",
|
|
55
|
+
default="",
|
|
56
|
+
)
|
|
57
|
+
parser.addini(
|
|
58
|
+
"querycount_duplicate_threshold",
|
|
59
|
+
help=(
|
|
60
|
+
"How many repeats of one query shape count as N+1 "
|
|
61
|
+
f"(default: {DEFAULT_DUPLICATE_THRESHOLD})."
|
|
62
|
+
),
|
|
63
|
+
default="",
|
|
64
|
+
)
|
|
65
|
+
|
|
66
|
+
|
|
67
|
+
def pytest_configure(config: pytest.Config) -> None:
|
|
68
|
+
config.addinivalue_line(
|
|
69
|
+
"markers",
|
|
70
|
+
"max_queries(n): fail the test if it runs more than n SQL queries.",
|
|
71
|
+
)
|
|
72
|
+
config.addinivalue_line(
|
|
73
|
+
"markers",
|
|
74
|
+
"no_n_plus_one(threshold=2, kinds=('select',)): fail the test if one "
|
|
75
|
+
"query shape is executed repeatedly.",
|
|
76
|
+
)
|
|
77
|
+
config.addinivalue_line(
|
|
78
|
+
"markers",
|
|
79
|
+
"no_seq_scan(ignore=()): fail the test if a query scans a table "
|
|
80
|
+
"sequentially because no index could serve its filter. PostgreSQL only.",
|
|
81
|
+
)
|
|
82
|
+
_STATS.clear()
|
|
83
|
+
backends.install_all()
|
|
84
|
+
|
|
85
|
+
|
|
86
|
+
def pytest_collection_modifyitems(
|
|
87
|
+
config: pytest.Config,
|
|
88
|
+
items: list[pytest.Item],
|
|
89
|
+
) -> None:
|
|
90
|
+
"""Reject malformed markers while collecting.
|
|
91
|
+
|
|
92
|
+
Validating here rather than mid-test means a typo is a clean usage error
|
|
93
|
+
instead of an exception raised from inside a hook wrapper.
|
|
94
|
+
"""
|
|
95
|
+
for item in items:
|
|
96
|
+
marker = item.get_closest_marker("max_queries")
|
|
97
|
+
if marker is not None:
|
|
98
|
+
_max_queries_from_marker(item, marker)
|
|
99
|
+
|
|
100
|
+
|
|
101
|
+
# -- per-test recording ----------------------------------------------------
|
|
102
|
+
|
|
103
|
+
|
|
104
|
+
@pytest.hookimpl(wrapper=True)
|
|
105
|
+
def pytest_runtest_call(item: pytest.Item) -> Generator[None, object, object]:
|
|
106
|
+
"""Record the call phase, then hold the test to its budget.
|
|
107
|
+
|
|
108
|
+
Deliberately wraps only the call phase: fixture setup and teardown run
|
|
109
|
+
outside it, so schema creation and seeding never consume a budget.
|
|
110
|
+
"""
|
|
111
|
+
__tracebackhide__ = True
|
|
112
|
+
|
|
113
|
+
budget = _budget_for(item)
|
|
114
|
+
recorder = Recorder(explain=budget.needs_explain)
|
|
115
|
+
item.stash[RECORDER_KEY] = recorder
|
|
116
|
+
backends.push(recorder)
|
|
117
|
+
try:
|
|
118
|
+
result = yield
|
|
119
|
+
except BaseException:
|
|
120
|
+
# The test itself failed. Record what we saw for the report, but let its
|
|
121
|
+
# exception through untouched: a budget complaint on top of a real
|
|
122
|
+
# failure buries the thing that actually needs fixing.
|
|
123
|
+
_record_stats(item, recorder)
|
|
124
|
+
raise
|
|
125
|
+
finally:
|
|
126
|
+
backends.pop(recorder)
|
|
127
|
+
|
|
128
|
+
_record_stats(item, recorder)
|
|
129
|
+
checks.enforce(budget, recorder, label=item.name)
|
|
130
|
+
return result
|
|
131
|
+
|
|
132
|
+
|
|
133
|
+
def _record_stats(item: pytest.Item, recorder: Recorder) -> None:
|
|
134
|
+
if not recorder.count:
|
|
135
|
+
return
|
|
136
|
+
duplicates = recorder.duplicates(threshold=2, kinds=DEFAULT_KINDS)
|
|
137
|
+
_STATS[item.nodeid] = report.TestStats(
|
|
138
|
+
nodeid=item.nodeid,
|
|
139
|
+
count=recorder.count,
|
|
140
|
+
duration=recorder.total_duration,
|
|
141
|
+
worst_duplicate=duplicates[0].count if duplicates else 0,
|
|
142
|
+
worst_sql=duplicates[0].sample.short_sql() if duplicates else None,
|
|
143
|
+
)
|
|
144
|
+
|
|
145
|
+
|
|
146
|
+
def _budget_for(item: pytest.Item) -> Budget:
|
|
147
|
+
"""Resolve the budget: marker first, then --querycount-max, then ini."""
|
|
148
|
+
budget = Budget(
|
|
149
|
+
max_queries=_default_max(item.config),
|
|
150
|
+
duplicate_threshold=_default_threshold(item.config),
|
|
151
|
+
no_seq_scan=bool(item.config.getoption("querycount_no_seq_scan")),
|
|
152
|
+
)
|
|
153
|
+
|
|
154
|
+
marker = item.get_closest_marker("max_queries")
|
|
155
|
+
if marker is not None:
|
|
156
|
+
budget.max_queries = _max_queries_from_marker(item, marker)
|
|
157
|
+
|
|
158
|
+
marker = item.get_closest_marker("no_seq_scan")
|
|
159
|
+
if marker is not None:
|
|
160
|
+
budget.no_seq_scan = True
|
|
161
|
+
budget.ignore_tables = tuple(marker.kwargs.get("ignore", ()))
|
|
162
|
+
|
|
163
|
+
marker = item.get_closest_marker("no_n_plus_one")
|
|
164
|
+
if marker is not None:
|
|
165
|
+
budget.no_n_plus_one = True
|
|
166
|
+
budget.duplicate_threshold = int(marker.kwargs.get("threshold", budget.duplicate_threshold))
|
|
167
|
+
if "kinds" in marker.kwargs:
|
|
168
|
+
kinds = marker.kwargs["kinds"]
|
|
169
|
+
budget.kinds = None if kinds is None else tuple(kinds)
|
|
170
|
+
|
|
171
|
+
return budget
|
|
172
|
+
|
|
173
|
+
|
|
174
|
+
def _max_queries_from_marker(item: pytest.Item, marker: pytest.Mark) -> int:
|
|
175
|
+
if marker.args:
|
|
176
|
+
value = marker.args[0]
|
|
177
|
+
elif "n" in marker.kwargs:
|
|
178
|
+
value = marker.kwargs["n"]
|
|
179
|
+
else:
|
|
180
|
+
raise pytest.UsageError(
|
|
181
|
+
f"{item.nodeid}: @pytest.mark.max_queries needs a number, "
|
|
182
|
+
"e.g. @pytest.mark.max_queries(3)"
|
|
183
|
+
)
|
|
184
|
+
try:
|
|
185
|
+
return int(value)
|
|
186
|
+
except (TypeError, ValueError):
|
|
187
|
+
raise pytest.UsageError(
|
|
188
|
+
f"{item.nodeid}: @pytest.mark.max_queries({value!r}) is not a number"
|
|
189
|
+
) from None
|
|
190
|
+
|
|
191
|
+
|
|
192
|
+
def _default_max(config: pytest.Config) -> int | None:
|
|
193
|
+
from_cli = config.getoption("querycount_max")
|
|
194
|
+
if from_cli is not None:
|
|
195
|
+
return int(from_cli)
|
|
196
|
+
from_ini = str(config.getini("querycount_max") or "").strip()
|
|
197
|
+
return int(from_ini) if from_ini else None
|
|
198
|
+
|
|
199
|
+
|
|
200
|
+
def _default_threshold(config: pytest.Config) -> int:
|
|
201
|
+
from_ini = str(config.getini("querycount_duplicate_threshold") or "").strip()
|
|
202
|
+
return int(from_ini) if from_ini else DEFAULT_DUPLICATE_THRESHOLD
|
|
203
|
+
|
|
204
|
+
|
|
205
|
+
# -- the fixture -----------------------------------------------------------
|
|
206
|
+
|
|
207
|
+
|
|
208
|
+
class _Block:
|
|
209
|
+
"""Context manager returned by ``querycount(...)``.
|
|
210
|
+
|
|
211
|
+
``__enter__`` yields the :class:`Recorder`, so the same object carries the
|
|
212
|
+
live count during the block and the evidence afterwards.
|
|
213
|
+
"""
|
|
214
|
+
|
|
215
|
+
def __init__(self, budget: Budget, label: str) -> None:
|
|
216
|
+
self._budget = budget
|
|
217
|
+
self._label = label
|
|
218
|
+
self.recorder = Recorder(explain=budget.needs_explain)
|
|
219
|
+
|
|
220
|
+
def __enter__(self) -> Recorder:
|
|
221
|
+
backends.push(self.recorder)
|
|
222
|
+
return self.recorder
|
|
223
|
+
|
|
224
|
+
def __exit__(self, exc_type: Any, exc: Any, tb: Any) -> None:
|
|
225
|
+
__tracebackhide__ = True
|
|
226
|
+
|
|
227
|
+
backends.pop(self.recorder)
|
|
228
|
+
# An exception from inside the block is the more important news.
|
|
229
|
+
if exc_type is None:
|
|
230
|
+
checks.enforce(self._budget, self.recorder, label=self._label)
|
|
231
|
+
|
|
232
|
+
|
|
233
|
+
@pytest.fixture
|
|
234
|
+
def querycount(request: pytest.FixtureRequest) -> Any:
|
|
235
|
+
"""Record the queries a block of code runs, and optionally budget them.
|
|
236
|
+
|
|
237
|
+
def test_detail(client, querycount):
|
|
238
|
+
with querycount(max_queries=2) as queries:
|
|
239
|
+
client.get("/users/1")
|
|
240
|
+
assert queries.duplicates() == []
|
|
241
|
+
|
|
242
|
+
Called with no arguments it only observes, which is the way to find out what
|
|
243
|
+
a budget should be before you commit to one.
|
|
244
|
+
"""
|
|
245
|
+
|
|
246
|
+
def factory(
|
|
247
|
+
max_queries: int | None = None,
|
|
248
|
+
no_n_plus_one: bool = False,
|
|
249
|
+
threshold: int = DEFAULT_DUPLICATE_THRESHOLD,
|
|
250
|
+
kinds: Sequence[str] | None = DEFAULT_KINDS,
|
|
251
|
+
no_seq_scan: bool = False,
|
|
252
|
+
ignore: Sequence[str] = (),
|
|
253
|
+
) -> _Block:
|
|
254
|
+
return _Block(
|
|
255
|
+
Budget(
|
|
256
|
+
max_queries=max_queries,
|
|
257
|
+
no_n_plus_one=no_n_plus_one,
|
|
258
|
+
duplicate_threshold=threshold,
|
|
259
|
+
kinds=kinds,
|
|
260
|
+
no_seq_scan=no_seq_scan,
|
|
261
|
+
ignore_tables=tuple(ignore),
|
|
262
|
+
),
|
|
263
|
+
label="This querycount block",
|
|
264
|
+
)
|
|
265
|
+
|
|
266
|
+
return factory
|
|
267
|
+
|
|
268
|
+
|
|
269
|
+
# -- summary ---------------------------------------------------------------
|
|
270
|
+
|
|
271
|
+
|
|
272
|
+
def pytest_terminal_summary(
|
|
273
|
+
terminalreporter: Any,
|
|
274
|
+
exitstatus: int,
|
|
275
|
+
config: pytest.Config,
|
|
276
|
+
) -> None:
|
|
277
|
+
if not config.getoption("querycount_report"):
|
|
278
|
+
return
|
|
279
|
+
lines = report.build(_STATS, top=int(config.getoption("querycount_top")))
|
|
280
|
+
if not lines:
|
|
281
|
+
return
|
|
282
|
+
terminalreporter.write_sep("=", "querycount summary")
|
|
283
|
+
for line in lines:
|
|
284
|
+
terminalreporter.write_line(line)
|
|
File without changes
|
|
@@ -0,0 +1,214 @@
|
|
|
1
|
+
"""Collecting queries and answering questions about them."""
|
|
2
|
+
|
|
3
|
+
from __future__ import annotations
|
|
4
|
+
|
|
5
|
+
import os
|
|
6
|
+
import sys
|
|
7
|
+
from collections import Counter
|
|
8
|
+
from collections.abc import Iterable, Sequence
|
|
9
|
+
from typing import TYPE_CHECKING
|
|
10
|
+
|
|
11
|
+
from pytest_querycount.records import Duplicate, QueryRecord
|
|
12
|
+
|
|
13
|
+
if TYPE_CHECKING:
|
|
14
|
+
from types import FrameType
|
|
15
|
+
|
|
16
|
+
# Frames belonging to these files are never the interesting caller: they are the
|
|
17
|
+
# ORM, the driver, the test runner, or us.
|
|
18
|
+
_INTERNAL_PATHS = (
|
|
19
|
+
"/sqlalchemy/",
|
|
20
|
+
"/pytest_querycount/",
|
|
21
|
+
"/_pytest/",
|
|
22
|
+
"/pluggy/",
|
|
23
|
+
"/asyncio/",
|
|
24
|
+
"/greenlet/",
|
|
25
|
+
"/contextlib.py",
|
|
26
|
+
"/threading.py",
|
|
27
|
+
)
|
|
28
|
+
|
|
29
|
+
_MAX_STACK_DEPTH = 60
|
|
30
|
+
|
|
31
|
+
|
|
32
|
+
def _shorten(path: str) -> str:
|
|
33
|
+
"""Relative to the working directory when that is shorter, absolute otherwise.
|
|
34
|
+
|
|
35
|
+
Failure messages are read in a terminal, and an absolute path to a pytest
|
|
36
|
+
temp directory pushes the useful part off the edge of the screen.
|
|
37
|
+
"""
|
|
38
|
+
try:
|
|
39
|
+
relative = os.path.relpath(path)
|
|
40
|
+
except ValueError: # pragma: no cover - different drive on Windows
|
|
41
|
+
return path
|
|
42
|
+
return path if relative.startswith("..") else relative
|
|
43
|
+
|
|
44
|
+
|
|
45
|
+
def caller_location(skip: int = 1) -> str | None:
|
|
46
|
+
"""``path:lineno`` of the closest frame that is not library machinery."""
|
|
47
|
+
try:
|
|
48
|
+
frame: FrameType | None = sys._getframe(skip + 1)
|
|
49
|
+
except ValueError: # pragma: no cover - stack shallower than `skip`
|
|
50
|
+
return None
|
|
51
|
+
depth = 0
|
|
52
|
+
while frame is not None and depth < _MAX_STACK_DEPTH:
|
|
53
|
+
filename = frame.f_code.co_filename.replace(os.sep, "/")
|
|
54
|
+
if not any(part in filename for part in _INTERNAL_PATHS):
|
|
55
|
+
return f"{_shorten(frame.f_code.co_filename)}:{frame.f_lineno}"
|
|
56
|
+
frame = frame.f_back
|
|
57
|
+
depth += 1
|
|
58
|
+
return None
|
|
59
|
+
|
|
60
|
+
|
|
61
|
+
class Recorder:
|
|
62
|
+
"""Accumulates :class:`QueryRecord` objects and reports on them.
|
|
63
|
+
|
|
64
|
+
A recorder only sees queries while it is on the active stack; see
|
|
65
|
+
:mod:`pytest_querycount.backends`.
|
|
66
|
+
"""
|
|
67
|
+
|
|
68
|
+
def __init__(self, explain: bool = False) -> None:
|
|
69
|
+
self.records: list[QueryRecord] = []
|
|
70
|
+
self.explain = explain
|
|
71
|
+
"""Whether the backend should obtain a query plan for each SELECT.
|
|
72
|
+
|
|
73
|
+
Off unless a check needs it: EXPLAIN means a second round trip per
|
|
74
|
+
query, which is a real cost to impose on a suite that did not ask."""
|
|
75
|
+
|
|
76
|
+
def add(self, record: QueryRecord) -> None:
|
|
77
|
+
self.records.append(record)
|
|
78
|
+
|
|
79
|
+
def clear(self) -> None:
|
|
80
|
+
self.records.clear()
|
|
81
|
+
|
|
82
|
+
# -- basic numbers ----------------------------------------------------
|
|
83
|
+
|
|
84
|
+
def __len__(self) -> int:
|
|
85
|
+
return len(self.records)
|
|
86
|
+
|
|
87
|
+
@property
|
|
88
|
+
def count(self) -> int:
|
|
89
|
+
"""How many statements were executed."""
|
|
90
|
+
return len(self.records)
|
|
91
|
+
|
|
92
|
+
@property
|
|
93
|
+
def total_duration(self) -> float:
|
|
94
|
+
"""Seconds spent in the driver, summed."""
|
|
95
|
+
return sum(record.duration for record in self.records)
|
|
96
|
+
|
|
97
|
+
def of_kind(self, *kinds: str) -> list[QueryRecord]:
|
|
98
|
+
wanted = {kind.lower() for kind in kinds}
|
|
99
|
+
return [record for record in self.records if record.kind in wanted]
|
|
100
|
+
|
|
101
|
+
# -- duplicate analysis ----------------------------------------------
|
|
102
|
+
|
|
103
|
+
@property
|
|
104
|
+
def fingerprints(self) -> Counter[str]:
|
|
105
|
+
return Counter(record.fingerprint for record in self.records)
|
|
106
|
+
|
|
107
|
+
def duplicates(
|
|
108
|
+
self,
|
|
109
|
+
threshold: int = 2,
|
|
110
|
+
kinds: Sequence[str] | None = ("select",),
|
|
111
|
+
) -> list[Duplicate]:
|
|
112
|
+
"""Query shapes that ran at least ``threshold`` times.
|
|
113
|
+
|
|
114
|
+
``kinds`` restricts the search to certain statement types; ``None``
|
|
115
|
+
considers all of them. The default looks only at SELECTs, because a
|
|
116
|
+
repeated SELECT is the N+1 signature, whereas repeated SAVEPOINTs and
|
|
117
|
+
BEGINs are simply how transactions work.
|
|
118
|
+
"""
|
|
119
|
+
candidates: Iterable[QueryRecord]
|
|
120
|
+
candidates = self.records if kinds is None else self.of_kind(*kinds)
|
|
121
|
+
|
|
122
|
+
counts: Counter[str] = Counter()
|
|
123
|
+
samples: dict[str, QueryRecord] = {}
|
|
124
|
+
locations: dict[str, Counter[str]] = {}
|
|
125
|
+
for record in candidates:
|
|
126
|
+
counts[record.fingerprint] += 1
|
|
127
|
+
samples.setdefault(record.fingerprint, record)
|
|
128
|
+
if record.location:
|
|
129
|
+
locations.setdefault(record.fingerprint, Counter())[record.location] += 1
|
|
130
|
+
|
|
131
|
+
found = [
|
|
132
|
+
Duplicate(
|
|
133
|
+
fingerprint=shape,
|
|
134
|
+
count=hits,
|
|
135
|
+
sample=samples[shape],
|
|
136
|
+
locations=locations.get(shape, Counter()),
|
|
137
|
+
)
|
|
138
|
+
for shape, hits in counts.items()
|
|
139
|
+
if hits >= threshold
|
|
140
|
+
]
|
|
141
|
+
found.sort(key=lambda duplicate: duplicate.count, reverse=True)
|
|
142
|
+
return found
|
|
143
|
+
|
|
144
|
+
# -- query plans ------------------------------------------------------
|
|
145
|
+
|
|
146
|
+
@property
|
|
147
|
+
def explained_count(self) -> int:
|
|
148
|
+
return sum(1 for record in self.records if record.explained)
|
|
149
|
+
|
|
150
|
+
def seq_scans(self, ignore: Sequence[str] = ()) -> list[QueryRecord]:
|
|
151
|
+
"""Records whose plan held a sequential scan no index could serve.
|
|
152
|
+
|
|
153
|
+
``ignore`` names tables to pass over -- a small lookup table is read
|
|
154
|
+
whole on purpose, and no index would improve it.
|
|
155
|
+
"""
|
|
156
|
+
skipped = {name.lower() for name in ignore}
|
|
157
|
+
return [
|
|
158
|
+
record
|
|
159
|
+
for record in self.records
|
|
160
|
+
if any(scan.relation.lower() not in skipped for scan in record.seq_scans)
|
|
161
|
+
]
|
|
162
|
+
|
|
163
|
+
# -- human output -----------------------------------------------------
|
|
164
|
+
|
|
165
|
+
def report(self, width: int = 100, limit: int | None = None, collapse: bool = True) -> str:
|
|
166
|
+
"""The recorded queries, in order, with timings and origins.
|
|
167
|
+
|
|
168
|
+
With ``collapse`` a run of identical shapes prints as one line -- twelve
|
|
169
|
+
copies of the same SELECT say nothing that ``12x`` does not, and they
|
|
170
|
+
push the useful lines off the screen.
|
|
171
|
+
"""
|
|
172
|
+
if not self.records:
|
|
173
|
+
return "no queries recorded"
|
|
174
|
+
|
|
175
|
+
groups = self._runs() if collapse else [[record] for record in self.records]
|
|
176
|
+
shown = groups if limit is None else groups[:limit]
|
|
177
|
+
|
|
178
|
+
lines = [
|
|
179
|
+
f"{self.count} quer{'y' if self.count == 1 else 'ies'} "
|
|
180
|
+
f"in {self.total_duration * 1000:.1f}ms"
|
|
181
|
+
]
|
|
182
|
+
position = 1
|
|
183
|
+
for group in shown:
|
|
184
|
+
first = group[0]
|
|
185
|
+
elapsed = sum(record.duration for record in group) * 1000
|
|
186
|
+
if len(group) == 1:
|
|
187
|
+
label = f"{position}."
|
|
188
|
+
timing = f"[{elapsed:7.2f}ms]"
|
|
189
|
+
else:
|
|
190
|
+
label = f"{position}-{position + len(group) - 1}."
|
|
191
|
+
timing = f"[{elapsed:7.2f}ms] {len(group)}x"
|
|
192
|
+
marker = " (executemany)" if first.executemany else ""
|
|
193
|
+
lines.append(f" {label:>8} {timing}{marker} {first.short_sql(width)}")
|
|
194
|
+
if first.location:
|
|
195
|
+
lines.append(f" {'':>8} {'':>10} from {first.location}")
|
|
196
|
+
position += len(group)
|
|
197
|
+
|
|
198
|
+
hidden = self.count - sum(len(group) for group in shown)
|
|
199
|
+
if hidden > 0:
|
|
200
|
+
lines.append(f" ... and {hidden} more")
|
|
201
|
+
return "\n".join(lines)
|
|
202
|
+
|
|
203
|
+
def _runs(self) -> list[list[QueryRecord]]:
|
|
204
|
+
"""Consecutive records sharing a fingerprint, grouped."""
|
|
205
|
+
runs: list[list[QueryRecord]] = []
|
|
206
|
+
for record in self.records:
|
|
207
|
+
if runs and runs[-1][0].fingerprint == record.fingerprint:
|
|
208
|
+
runs[-1].append(record)
|
|
209
|
+
else:
|
|
210
|
+
runs.append([record])
|
|
211
|
+
return runs
|
|
212
|
+
|
|
213
|
+
def __repr__(self) -> str:
|
|
214
|
+
return f"<Recorder {self.count} queries, {self.total_duration * 1000:.1f}ms>"
|
|
@@ -0,0 +1,66 @@
|
|
|
1
|
+
"""The data the recorder collects."""
|
|
2
|
+
|
|
3
|
+
from __future__ import annotations
|
|
4
|
+
|
|
5
|
+
from collections import Counter
|
|
6
|
+
from dataclasses import dataclass, field
|
|
7
|
+
|
|
8
|
+
from pytest_querycount.explain import SeqScan
|
|
9
|
+
|
|
10
|
+
|
|
11
|
+
@dataclass(frozen=True)
|
|
12
|
+
class QueryRecord:
|
|
13
|
+
"""One statement handed to the driver."""
|
|
14
|
+
|
|
15
|
+
sql: str
|
|
16
|
+
"""The statement as the driver saw it, literals and placeholders intact."""
|
|
17
|
+
|
|
18
|
+
fingerprint: str
|
|
19
|
+
"""``sql`` with its literals replaced -- see :func:`~.normalize.fingerprint`."""
|
|
20
|
+
|
|
21
|
+
kind: str
|
|
22
|
+
"""Leading keyword, lowercased: ``"select"``, ``"insert"``, ..."""
|
|
23
|
+
|
|
24
|
+
duration: float
|
|
25
|
+
"""Seconds spent inside the driver call."""
|
|
26
|
+
|
|
27
|
+
executemany: bool
|
|
28
|
+
"""True when one call carried many parameter sets. Counts as a single query,
|
|
29
|
+
because it is a single round trip -- which is exactly why it is the fix for
|
|
30
|
+
a loop of inserts."""
|
|
31
|
+
|
|
32
|
+
rowcount: int | None = None
|
|
33
|
+
|
|
34
|
+
seq_scans: tuple[SeqScan, ...] = ()
|
|
35
|
+
"""Sequential scans the planner kept even with seq scans penalised, i.e. ones
|
|
36
|
+
no index could serve. Only populated when a check asked for EXPLAIN."""
|
|
37
|
+
|
|
38
|
+
explained: bool = False
|
|
39
|
+
"""Whether a plan was actually obtained. Distinguishes "no problems found"
|
|
40
|
+
from "we never looked", which must not be reported the same way."""
|
|
41
|
+
|
|
42
|
+
location: str | None = None
|
|
43
|
+
"""``path:lineno`` of the nearest frame outside the ORM and the test runner,
|
|
44
|
+
i.e. the line of *your* code that caused this."""
|
|
45
|
+
|
|
46
|
+
def short_sql(self, width: int = 100) -> str:
|
|
47
|
+
collapsed = " ".join(self.sql.split())
|
|
48
|
+
if len(collapsed) <= width:
|
|
49
|
+
return collapsed
|
|
50
|
+
return collapsed[: width - 1] + "…"
|
|
51
|
+
|
|
52
|
+
|
|
53
|
+
@dataclass(frozen=True)
|
|
54
|
+
class Duplicate:
|
|
55
|
+
"""A query shape that ran more than once."""
|
|
56
|
+
|
|
57
|
+
fingerprint: str
|
|
58
|
+
count: int
|
|
59
|
+
sample: QueryRecord
|
|
60
|
+
locations: Counter[str] = field(default_factory=Counter)
|
|
61
|
+
|
|
62
|
+
def describe(self, width: int = 100) -> str:
|
|
63
|
+
lines = [f"{self.count}x {self.sample.short_sql(width)}"]
|
|
64
|
+
for location, hits in self.locations.most_common(3):
|
|
65
|
+
lines.append(f" from {location} ({hits}x)")
|
|
66
|
+
return "\n".join(lines)
|