attestql 0.2.1__py3-none-any.whl

This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
attestql/__init__.py ADDED
@@ -0,0 +1,6 @@
1
+ """AttestQL: audit text-to-SQL gold and predictions on PostgreSQL with typed replay evidence.
2
+
3
+ ``audit`` is the command and its backend, ``evidence`` the record, serializer and comparator it
4
+ writes and reads, ``kernel`` the shared result types. ADR-0013 records the decision this package
5
+ implements.
6
+ """
@@ -0,0 +1,82 @@
1
+ """The audit: two statements, one database, one verdict with the evidence behind it.
2
+
3
+ ``parse`` says what an audit reads off a statement and ``statements`` is PostgreSQL's
4
+ answer to it, ``backend`` says what an audit needs from a database and ``postgres`` is
5
+ the first one that gives it, ``engines`` names one engine's answer to both, ``fixture``
6
+ measures the data both statements read, ``compare`` runs them and states what it found,
7
+ and ``smells`` asks the database the four questions that make a gold worth reading again.
8
+
9
+ ``engines`` is not imported here either, and for the same reason as ``cli``: it names the
10
+ PostgreSQL backend, so importing it would pull the driver into every import of this
11
+ package.
12
+
13
+ ``cli`` is deliberately not imported here. It builds the PostgreSQL backend, so importing
14
+ it would pull the driver into every import of this package, including the ones a test
15
+ makes to drive a scripted backend that reaches no server at all.
16
+ """
17
+
18
+ from attestql.audit.backend import (
19
+ Backend,
20
+ BackendRefused,
21
+ ReadBackDrift,
22
+ ShuffledCopies,
23
+ StatementTimedOut,
24
+ TextCensus,
25
+ )
26
+ from attestql.audit.compare import (
27
+ BirdEx,
28
+ Comparison,
29
+ RecordedStatement,
30
+ TestSuiteEx,
31
+ bird_ex,
32
+ compare_statements,
33
+ counterexample_json,
34
+ record_statement,
35
+ test_suite_ex,
36
+ write_comparison,
37
+ )
38
+ from attestql.audit.fixture import fixture_digest
39
+ from attestql.audit.parse import (
40
+ OrderingKey,
41
+ ParsedStatement,
42
+ ParserIdentity,
43
+ StatementRefused,
44
+ )
45
+ from attestql.audit.smells import (
46
+ QuestionText,
47
+ Smell,
48
+ SmellSettings,
49
+ all_smells,
50
+ smells_json,
51
+ )
52
+ from attestql.audit.statements import parse_statement
53
+
54
+ __all__ = [
55
+ "Backend",
56
+ "BackendRefused",
57
+ "BirdEx",
58
+ "Comparison",
59
+ "OrderingKey",
60
+ "ParsedStatement",
61
+ "ParserIdentity",
62
+ "QuestionText",
63
+ "ReadBackDrift",
64
+ "RecordedStatement",
65
+ "ShuffledCopies",
66
+ "Smell",
67
+ "SmellSettings",
68
+ "StatementRefused",
69
+ "StatementTimedOut",
70
+ "TestSuiteEx",
71
+ "TextCensus",
72
+ "all_smells",
73
+ "bird_ex",
74
+ "compare_statements",
75
+ "counterexample_json",
76
+ "fixture_digest",
77
+ "parse_statement",
78
+ "record_statement",
79
+ "smells_json",
80
+ "test_suite_ex",
81
+ "write_comparison",
82
+ ]
@@ -0,0 +1,453 @@
1
+ """What an audit needs from a database, stated without naming one.
2
+
3
+ ADR-0013 point 4: PostgreSQL first, because typed replay needs real column types, and
4
+ SQLite next, because the original BIRD runs on it. So the interface is engine-neutral
5
+ from the first commit and nothing in it is PostgreSQL's: no type oid, no search path, no
6
+ transaction syntax. A backend is asked for typed rows, for the settings that decide
7
+ whether two executions are comparable, and for digests of the data that was read; how it
8
+ gets them is its own.
9
+
10
+ Read-only is the backend's promise and not the caller's request. ``execute`` takes SQL
11
+ and a timeout and nothing that could make it write, and an implementation that cannot
12
+ prove to itself that the statement ran read-only raises ``ReadBackDrift`` rather than
13
+ returning rows whose provenance it cannot state. That is the one property ADR-0013 point
14
+ 7 carried over from the deleted executor, and it lives here at the interface so a second
15
+ engine cannot quietly drop it.
16
+
17
+ The digests are three calls rather than one so that the expensive one is optional.
18
+ ``schema_digest`` and ``row_counts`` are always taken; ``content_digests`` reads every
19
+ row of every table and is taken only when the caller asks for it (ADR-0013 point 6).
20
+ ``content_signal`` is none of the three: it reads what the engine already counts about
21
+ its own tables, so that a measurement cached by an earlier run can be told apart from one
22
+ the data has moved under since, and from one taken before an analyze the planner has since
23
+ chosen a different plan from. ``planner_statistics`` reads the same catalogue for what a
24
+ record and a summary state: what the plan of every rerun this run makes was chosen from. ``existing_tables`` is asked before any of them, because a name the data does not hold is
25
+ one question's error and never the end of a run. It answers in two parts, because there are
26
+ two ways a name can fail to be measurable and they are repaired in different places: a table
27
+ nobody loaded is a defect in the question file, and a table the audit's login was never
28
+ granted is a defect in the grants.
29
+ """
30
+
31
+ from __future__ import annotations
32
+
33
+ from collections.abc import Mapping, Sequence
34
+ from dataclasses import dataclass
35
+ from string import ascii_lowercase, ascii_uppercase
36
+ from typing import Protocol
37
+
38
+ from attestql.evidence.render import Json
39
+ from attestql.evidence.types import SessionSettings
40
+ from attestql.kernel.types import ExecutionResult
41
+
42
+ ASCII_LETTERS_FOLDED = str.maketrans(ascii_uppercase, ascii_lowercase)
43
+ """The 26 letters and nothing else, which is the whole of the rule below."""
44
+
45
+
46
+ def folded(name: str) -> str:
47
+ """One name as an engine matches two spellings of it: the ASCII letters, and no more.
48
+
49
+ Both engines fold exactly that much. SQLite compares an identifier with an ASCII
50
+ case-insensitive comparison and leaves every other character as it is, so ``STRASSE`` and
51
+ ``straße`` are two names there; PostgreSQL downcases an unquoted identifier and leaves a
52
+ multibyte character alone.
53
+
54
+ Python's ``casefold`` is the Unicode rule and is wider than either: it folds ``ß`` to
55
+ ``ss``. Matching a name with it makes this tool read a statement the way no engine does,
56
+ which is how a sort key that names no column comes to be read as naming one, and how a
57
+ table the file does not hold comes to be measured as one it does.
58
+ """
59
+ return name.translate(ASCII_LETTERS_FOLDED)
60
+
61
+
62
+ @dataclass(frozen=True, order=True)
63
+ class TableName:
64
+ """One relation as a statement named it: the schema it wrote, and the relation.
65
+
66
+ The two are kept apart because a name is not a string with a dot in it. A relation
67
+ called ``"a.b"`` is one name, and the table ``b`` of a schema ``a`` is another; a
68
+ string that joined them would be the same string for both, and whatever split it back
69
+ would have to guess. So the parse hands the two parts over as it read them and nothing
70
+ downstream takes ``text`` apart again.
71
+
72
+ ``schema`` is the empty string when the statement named none. What an unqualified name
73
+ means is the backend's to decide, because the answer is where the engine looks: a
74
+ caller that filled it in here would be stating one engine's default as the parse.
75
+ """
76
+
77
+ schema: str
78
+ name: str
79
+
80
+ @property
81
+ def text(self) -> str:
82
+ """The name as a summary and a record print it, qualified where the statement was.
83
+
84
+ For display and for keying, never for parsing back: whoever needs the parts has
85
+ them beside this.
86
+ """
87
+ return f"{self.schema}.{self.name}" if self.schema else self.name
88
+
89
+
90
+ class BackendRefused(RuntimeError):
91
+ """The backend will not produce a result it cannot state the provenance of.
92
+
93
+ ``step`` names what refused, so a caller reports which of the audit's own
94
+ preconditions was not met rather than a driver's message about a socket.
95
+ """
96
+
97
+ def __init__(self, step: str, detail: str) -> None:
98
+ self.step = step
99
+ self.detail = detail
100
+ super().__init__(f"{step}: {detail}")
101
+
102
+
103
+ class StatementTimedOut(BackendRefused):
104
+ """The statement ran past the bound this run put on it, and nothing else refused it.
105
+
106
+ Its own type because it is a different finding from every other refusal and the summary
107
+ counts it apart: a refusal is the database's answer about the statement, and this is the
108
+ run's own budget running out over one that was still running, which says as much about
109
+ the bound the operator chose as about the statement. A reader who sees the count on the
110
+ summary line knows a longer ``--statement-timeout`` would have compared those questions,
111
+ where a refused statement would have been refused at any bound.
112
+
113
+ ``step`` is always ``execute``, because reaching the bound is the statement running, and
114
+ ``detail`` is the backend's own words unchanged, so the line a reader is shown is what
115
+ the engine said and no error message moves for this distinction existing. ``seconds`` is
116
+ the bound that ran out, kept so that whoever reports it need not read it back out of a
117
+ message an engine wrote.
118
+ """
119
+
120
+ def __init__(self, seconds: int, detail: str) -> None:
121
+ self.seconds = seconds
122
+ super().__init__("execute", detail)
123
+
124
+
125
+ class ReadBackDrift(BackendRefused):
126
+ """The session did not hold what the executor set on it.
127
+
128
+ Raised instead of returning the rows. A result produced under a session whose
129
+ read-only flag or timeout is not what was asked for is a result about a different
130
+ execution than the one the record would describe, and the audit has no use for it.
131
+ """
132
+
133
+
134
+ @dataclass(frozen=True)
135
+ class TextCensus:
136
+ """How many of a text column's values look like numbers, counted on the server.
137
+
138
+ One question, four counts, one pass over the column: the rows it has, the ones that
139
+ are null, the ones that are empty, and the ones that are neither and still do not
140
+ match the pattern. A smell that orders by such a column asks this before it claims
141
+ the ordering is lexicographic over numbers, and the counts are its evidence.
142
+ """
143
+
144
+ rows: int
145
+ nulls: int
146
+ empty_strings: int
147
+ non_numeric: int
148
+ pattern: str
149
+
150
+
151
+ @dataclass(frozen=True)
152
+ class TableLookup:
153
+ """Of the names asked about, the ones the database holds and the ones it will not read.
154
+
155
+ ``present`` exists and this login may read it, which is what a fixture measurement can
156
+ cover. ``unreadable`` exists and the login may not read it, which is a grant nobody made
157
+ rather than a table nobody loaded; a name in neither is absent. Two states, two words,
158
+ because the operator repairs them in two different places and a summary that spelled them
159
+ the same way sent them to the wrong one.
160
+
161
+ The names are the caller's spelling in the caller's order and without duplicates, because
162
+ the caller is what has to say which of the names it asked about ended up where.
163
+ """
164
+
165
+ present: tuple[TableName, ...]
166
+ unreadable: tuple[TableName, ...]
167
+
168
+
169
+ @dataclass(frozen=True)
170
+ class PlannerStatistics:
171
+ """When one table's planner statistics were last taken, and how far its rows have moved.
172
+
173
+ The plan a statement is read with is chosen from these, and a probe that reruns a gold
174
+ and compares the two answers is comparing two plans whenever they moved in between. So
175
+ they are recorded beside what was measured under them. They are never repaired here:
176
+ the audit reads, and ANALYZE writes.
177
+
178
+ The timestamps are the engine's own rendering of them, or ``None`` where it holds none,
179
+ which is what a table nobody has analysed since the statistics were last reset says.
180
+ ``n_mod_since_analyze`` is what the engine counts as changed since the last one.
181
+ """
182
+
183
+ last_analyze: str | None
184
+ last_autoanalyze: str | None
185
+ n_mod_since_analyze: int
186
+
187
+
188
+ def planner_statistics_json(statistics: Mapping[TableName, PlannerStatistics]) -> Json:
189
+ """Those statistics as a summary and a smell's evidence write them, under the names asked.
190
+
191
+ One rendering because two runs are compared by diffing what they wrote: a block in a
192
+ record and a block in a summary that spelled the same three fields differently would be
193
+ two things to read instead of one.
194
+ """
195
+ return {
196
+ name.text: {
197
+ "last_analyze": measured.last_analyze,
198
+ "last_autoanalyze": measured.last_autoanalyze,
199
+ "n_mod_since_analyze": measured.n_mod_since_analyze,
200
+ }
201
+ for name, measured in statistics.items()
202
+ }
203
+
204
+
205
+ @dataclass(frozen=True)
206
+ class ShuffledCopies:
207
+ """What a shuffled copy of the data covers, and what it does not.
208
+
209
+ ``copied`` are the tables a rerun will read instead of the originals; ``skipped``
210
+ names the tables left behind with the count that made them too large, and
211
+ ``unreachable`` the ones a rerun would go on reading in place whatever was copied,
212
+ each with the reason. A smell that reruns a statement says which part of the data it
213
+ did not shuffle, and the three fields are what it says it from.
214
+
215
+ The names are the caller's own, so a table a gold named twice in two spellings is
216
+ answered under each of them: whether a rerun reaches a copy is a property of how the
217
+ statement wrote the name and not of the table underneath it.
218
+ """
219
+
220
+ copied: tuple[TableName, ...]
221
+ skipped: Mapping[TableName, int]
222
+ unreachable: Mapping[TableName, str]
223
+ seed: str
224
+ row_limit: int
225
+
226
+
227
+ class Backend(Protocol):
228
+ """One database, read-only, for the length of an audit."""
229
+
230
+ @property
231
+ def scratch(self) -> str:
232
+ """Where the shuffled copies of this run live, in the engine's own words.
233
+
234
+ Read off the backend and not off the options, because the option is what the run
235
+ asked for and this is what the engine made of it: a schema the login already holds
236
+ on one engine, and the connection's own TEMP database on one that needs nothing
237
+ arranged. A summary states it beside the seed and the row limit, so a reader is told
238
+ where a rerun's rows came from rather than what the command line said.
239
+ """
240
+ raise NotImplementedError
241
+
242
+ def identity(self) -> str:
243
+ """Engine, version, host or path, and database, as one line.
244
+
245
+ Two records that name different backends are still compared: the identity is
246
+ evidence a reader needs, not a precondition. It is one string because that is
247
+ what a record can state about any engine without a field per engine.
248
+ """
249
+ raise NotImplementedError
250
+
251
+ def effective_database_role(self) -> str:
252
+ """The identity the statements run as, which decides what they could read."""
253
+ raise NotImplementedError
254
+
255
+ def session_settings(self) -> SessionSettings:
256
+ """The seven settings that decide comparability, and everything else read back.
257
+
258
+ Asked once for a whole run, like ``existing_tables``: the summary and every record
259
+ state one read of the session, so that two records of one run cannot say the
260
+ statements behind them ran under two different sessions.
261
+
262
+ One block answers for either engine. A backend whose engine has no session states
263
+ the seven as absent and records what its engine can be asked about itself, so a
264
+ reader of the record reads the same shape whichever backend produced it.
265
+ """
266
+ raise NotImplementedError
267
+
268
+ def execute(self, sql: str, *, statement_timeout_seconds: int) -> ExecutionResult:
269
+ """Run one statement read-only under that timeout and return every row it gave.
270
+
271
+ Raises ``BackendRefused`` when the statement could not be run as stated and
272
+ ``ReadBackDrift`` when the session did not hold the envelope it was given.
273
+ """
274
+ raise NotImplementedError
275
+
276
+ def existing_tables(self, tables: Sequence[TableName]) -> TableLookup:
277
+ """Which of those names the database holds and may read, in the spelling given.
278
+
279
+ A gold that names a table this database does not have is one question's error and
280
+ not the end of a run, so what is measured before the questions asks this first
281
+ rather than discovering it by failing: what exists and can be read is measured, the
282
+ rest is named in the summary, and the questions that reference it fail on their own
283
+ lines with the server's own message.
284
+
285
+ Existence and readability are one question here and not two, because an
286
+ implementation that asked them apart could answer them of two different moments,
287
+ and because the answer is asked once for a whole run.
288
+ """
289
+ raise NotImplementedError
290
+
291
+ def schema_digest(self, tables: Sequence[TableName]) -> str:
292
+ """One digest over the table, column, type and nullability of those tables."""
293
+ raise NotImplementedError
294
+
295
+ def row_counts(self, tables: Sequence[TableName]) -> Mapping[str, int]:
296
+ """The exact number of rows in each table, counted rather than estimated."""
297
+ raise NotImplementedError
298
+
299
+ def content_digests(self, tables: Sequence[TableName]) -> Mapping[str, str]:
300
+ """A digest of the sorted rows of each table. The expensive one, asked for by name."""
301
+ raise NotImplementedError
302
+
303
+ def content_signal(self, tables: Sequence[TableName]) -> Mapping[str, str]:
304
+ """One cheap string per qualified table that moves when the table's rows move.
305
+
306
+ Read from whatever the engine already counts about its own tables, so that asking
307
+ it costs one question for a whole run rather than a pass over the data. It is not
308
+ a digest and never appears in a record: it exists so that a cached measurement of
309
+ data that has since changed can be recognised as stale and taken again.
310
+ """
311
+ raise NotImplementedError
312
+
313
+ def planner_statistics(
314
+ self, tables: Sequence[TableName]
315
+ ) -> Mapping[TableName, PlannerStatistics]:
316
+ """Per table, when the planner's statistics for it were last taken and how far off.
317
+
318
+ Asked because the shuffle probe reruns a gold over copies of the data and reads
319
+ both with whatever plan the planner chose, and the planner chooses from these. A
320
+ probe that fires on one run and is quiet on the next over the same data has this
321
+ underneath it, and a run that did not record them leaves a reader to guess.
322
+
323
+ Answered under the caller's own names, like ``column_types``, and a name the engine
324
+ keeps no statistics for gets no entry rather than an invented zero: an engine that
325
+ counts nothing of the kind answers with nothing at all.
326
+ """
327
+ raise NotImplementedError
328
+
329
+ def column_types(self, tables: Sequence[TableName]) -> Mapping[TableName, Mapping[str, str]]:
330
+ """Per table, the declared type of each of its columns.
331
+
332
+ Asked because an ordering key that resolves to a column is only interesting to a
333
+ smell when the column is declared as text: what is being looked for is an
334
+ ordering that is lexicographic where the question means numeric, and the
335
+ statement alone cannot say which one it is.
336
+
337
+ Answered under the caller's own names, unlike the digests above, because the
338
+ caller resolves a key against the name its statement wrote: two tables of one bare
339
+ name in two schemas are two entries here, and a lookup that had to match them by
340
+ the tail would read either one.
341
+ """
342
+ raise NotImplementedError
343
+
344
+ def declared_type_is_text(self, declared_type: str) -> bool:
345
+ """Whether a column declared that way holds text, as this engine reads a declaration.
346
+
347
+ Asked beside ``column_types`` and answered by the engine, because a declaration is
348
+ read by the engine's own rule and not by the word it is spelled with: one catalogue
349
+ renders a fixed set of type names and settles it by the name, and another keeps the
350
+ text of the CREATE statement and settles it by what that text contains, so ``VARCHAR
351
+ (50)`` is a text column there and matches no name at all.
352
+
353
+ A declaration this cannot place is not text, which is the conservative direction: a
354
+ smell that reads it stays quiet rather than claiming an ordering over numbers.
355
+ """
356
+ raise NotImplementedError
357
+
358
+ def order_sensitive_aggregate_types(self) -> frozenset[str]:
359
+ """The result column types whose aggregates depend on the order the rows were read.
360
+
361
+ Asked because a probe that reruns a gold over the same rows in another physical
362
+ order has to decide what a changed cell means, and for one class of value it means
363
+ nothing: a floating type whose aggregate is added up value by value gives another
364
+ last digit when the values arrive in another order, and that is arithmetic and not
365
+ a property of the statement. Every other changed cell is the statement depending on
366
+ the storage order, which is the finding.
367
+
368
+ Which types those are is the engine's answer and not the probe's, because it is a
369
+ property of how the engine adds: an engine that sums with a compensation gives the
370
+ same total whatever order it reads in, and answers with the empty set. The empty set
371
+ is the conservative one: every changed cell is then reported under the stronger name.
372
+ """
373
+ raise NotImplementedError
374
+
375
+ def numeric_text_census(self, table: TableName, column: str, pattern: str) -> TextCensus:
376
+ """Count that column's rows, nulls, empties and values the pattern rejects.
377
+
378
+ ``pattern`` is a POSIX regular expression. It is the caller's, so what counts as
379
+ a numeric-looking value is stated once, in the smell that decides it, rather than
380
+ once per engine; an engine whose pattern dialect differs translates it here.
381
+ """
382
+ raise NotImplementedError
383
+
384
+ def prepare_shuffled_copies(
385
+ self, tables: Sequence[TableName], *, seed: str, row_limit: int
386
+ ) -> ShuffledCopies:
387
+ """Copy those tables into scratch storage in a seeded order, once for a run.
388
+
389
+ A copy holds the same rows in another physical order, so a statement rerun
390
+ against it answers whether the result was a function of the data or of the order
391
+ the rows happened to be stored in. The order is seeded, so a run reproduces. A
392
+ table with more rows than ``row_limit`` is skipped and named in the answer rather
393
+ than copied, and so is a name a rerun would not reach the copy of however it was
394
+ made: the answer says which tables a rerun really reads differently, because a
395
+ smell that assumed the rest reports a statement as surviving a shuffle it never
396
+ had.
397
+
398
+ The scratch storage exists before the run and is not made here: an implementation
399
+ writes its copies into a place the login already holds and creates no container of
400
+ its own, because the audit's login is a read-only one and a tool that needed the
401
+ right to create one could not be run by the people this is for. Scratch storage
402
+ that is missing or that the login cannot write to is a ``BackendRefused`` naming
403
+ which of the two it was, and the caller reports the shuffle as not run rather than
404
+ stopping the audit.
405
+ """
406
+ raise NotImplementedError
407
+
408
+ def drop_shuffled_copies(self) -> None:
409
+ """Remove the copies this run made, and nothing else it did not.
410
+
411
+ Called in a finally, and safe when there are none. The scratch storage itself
412
+ outlives the run: it was arranged for the login and may hold another run's work,
413
+ so what is removed is what this run created in it.
414
+ """
415
+ raise NotImplementedError
416
+
417
+ def execute_shuffled(self, sql: str, *, statement_timeout_seconds: int) -> ExecutionResult:
418
+ """Run one statement read-only against the shuffled copies rather than the tables."""
419
+ raise NotImplementedError
420
+
421
+ def execute_plan_variant(self, sql: str, *, statement_timeout_seconds: int) -> ExecutionResult:
422
+ """Run one statement read-only with the engine steered away from its chosen plan.
423
+
424
+ The same rows read another way. What the steering is belongs to the engine: the
425
+ interface asks for another plan over the same data and does not name a setting.
426
+ """
427
+ raise NotImplementedError
428
+
429
+ def default_collation(self) -> str:
430
+ """The database's default collation, which decides what ordering text means.
431
+
432
+ Read separately from ``session_settings`` because it is a property of the
433
+ database and not of the session, and stated inside it because a record compares
434
+ one value: an implementation reads it here and puts it there, so the two can
435
+ never disagree.
436
+ """
437
+ raise NotImplementedError
438
+
439
+
440
+ __all__ = [
441
+ "ASCII_LETTERS_FOLDED",
442
+ "Backend",
443
+ "BackendRefused",
444
+ "PlannerStatistics",
445
+ "ReadBackDrift",
446
+ "ShuffledCopies",
447
+ "StatementTimedOut",
448
+ "TableLookup",
449
+ "TableName",
450
+ "TextCensus",
451
+ "folded",
452
+ "planner_statistics_json",
453
+ ]