sqlitexplorer 1.0.0__py3-none-any.whl

This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
sqlitexplorer/core.py ADDED
@@ -0,0 +1,700 @@
1
+ """SQLite access layer of sqlitexplorer.
2
+
3
+ The CLI never touches :mod:`sqlite3` directly: it opens a database with
4
+ :func:`open_database` and talks to the resulting :class:`Explorer`. Anything
5
+ that should reach the user as a plain message is raised as
6
+ :class:`ExplorerError`. Nothing in this module prints.
7
+ """
8
+
9
+ from __future__ import annotations
10
+
11
+ import sqlite3
12
+ from collections.abc import Iterable, Iterator, Mapping, Sequence
13
+ from contextlib import contextmanager
14
+ from dataclasses import dataclass, field
15
+ from itertools import islice
16
+ from pathlib import Path
17
+ from typing import NamedTuple
18
+
19
+ __all__ = [
20
+ "Explorer",
21
+ "ExplorerError",
22
+ "Page",
23
+ "ReadOnlyError",
24
+ "ResultSet",
25
+ "RowStream",
26
+ "StatsReport",
27
+ "open_database",
28
+ "plan_tree",
29
+ "quote_identifier",
30
+ "split_statements",
31
+ "translate_error",
32
+ ]
33
+
34
+ Parameters = Sequence[object] | Mapping[str, object]
35
+
36
+
37
+ class ExplorerError(Exception):
38
+ """An error that should be shown to the user as a plain message."""
39
+
40
+
41
+ class ReadOnlyError(ExplorerError):
42
+ """A write was attempted on a database that was opened read-only."""
43
+
44
+
45
+ class Page(NamedTuple):
46
+ """A window of rows: 1-based page ``number`` of ``size`` rows each."""
47
+
48
+ number: int
49
+ size: int
50
+
51
+ @property
52
+ def start(self) -> int:
53
+ return (self.number - 1) * self.size
54
+
55
+
56
+ @dataclass(frozen=True, slots=True)
57
+ class ResultSet:
58
+ """Outcome of one SQL statement.
59
+
60
+ ``columns`` is empty for statements that do not return rows (DDL, DML).
61
+ ``rowcount`` is the number of rows changed by DML, or -1 when it does not
62
+ apply. When the rows are only one :class:`Page` of the result, ``total``
63
+ tells how many rows the statement produced in all.
64
+ """
65
+
66
+ columns: tuple[str, ...] = ()
67
+ rows: list[tuple] = field(default_factory=list)
68
+ rowcount: int = -1
69
+ total: int | None = None
70
+
71
+ @property
72
+ def returns_rows(self) -> bool:
73
+ """Whether the statement produced a result set, even an empty one."""
74
+ return bool(self.columns)
75
+
76
+
77
+ @dataclass(slots=True)
78
+ class RowStream:
79
+ """Rows of a statement as they come out of the cursor, without loading them all."""
80
+
81
+ columns: tuple[str, ...] = ()
82
+ rows: Iterator[tuple] = field(default_factory=lambda: iter(()))
83
+ rowcount: int = -1
84
+
85
+ @property
86
+ def returns_rows(self) -> bool:
87
+ return bool(self.columns)
88
+
89
+ def collect(self) -> ResultSet:
90
+ """Load every remaining row into a :class:`ResultSet`."""
91
+ return ResultSet(columns=self.columns, rows=list(self.rows), rowcount=self.rowcount)
92
+
93
+
94
+ class StatsReport(NamedTuple):
95
+ """Per-column statistics, plus how many rows they were computed on."""
96
+
97
+ result: ResultSet
98
+ rows: int
99
+ sampled: bool
100
+
101
+
102
+ def _rows_of(cursor: sqlite3.Cursor) -> Iterator[tuple]:
103
+ try:
104
+ yield from cursor
105
+ finally:
106
+ cursor.close()
107
+
108
+
109
+ def _window(stream: RowStream, page: Page) -> ResultSet:
110
+ """Keep only *page* of *stream*, counting the rest without retaining it."""
111
+ skipped = sum(1 for _ in islice(stream.rows, page.start))
112
+ rows = list(islice(stream.rows, page.size))
113
+ rest = sum(1 for _ in stream.rows)
114
+ return ResultSet(
115
+ columns=stream.columns,
116
+ rows=rows,
117
+ rowcount=stream.rowcount,
118
+ total=skipped + len(rows) + rest,
119
+ )
120
+
121
+
122
+ def quote_identifier(name: str) -> str:
123
+ """Quote *name* so it can be embedded in SQL as an identifier."""
124
+ return '"' + name.replace('"', '""') + '"'
125
+
126
+
127
+ def split_statements(sql: str) -> list[str]:
128
+ """Split *sql* into complete statements.
129
+
130
+ Uses :func:`sqlite3.complete_statement`, so semicolons inside strings,
131
+ comments and ``CREATE TRIGGER ... END`` blocks do not split. Trailing text
132
+ without a semicolon is returned as a final statement.
133
+ """
134
+ statements: list[str] = []
135
+ buffer = ""
136
+ *parts, remainder = sql.split(";")
137
+ for part in parts:
138
+ buffer += part + ";"
139
+ if sqlite3.complete_statement(buffer):
140
+ if _has_content(buffer):
141
+ statements.append(buffer.strip())
142
+ buffer = ""
143
+ leftover = buffer + remainder
144
+ if _has_content(leftover):
145
+ statements.append(leftover.strip())
146
+ return statements
147
+
148
+
149
+ def _has_content(text: str) -> bool:
150
+ return bool(text.strip().rstrip(";").strip())
151
+
152
+
153
+ def plan_tree(result: ResultSet) -> ResultSet:
154
+ """Indent the ``detail`` column of an ``EXPLAIN QUERY PLAN`` result by depth."""
155
+ parents = {row[0]: row[1] for row in result.rows}
156
+
157
+ def depth(node: object) -> int:
158
+ level, seen = 0, set()
159
+ while node in parents and node not in seen and parents[node]:
160
+ seen.add(node)
161
+ node = parents[node]
162
+ level += 1
163
+ return level
164
+
165
+ rows = [(row[0], row[1], " " * depth(row[0]) + str(row[3])) for row in result.rows]
166
+ return ResultSet(columns=("id", "parent", "detail"), rows=rows)
167
+
168
+
169
+ def _database_uri(path: Path, *, write: bool) -> str:
170
+ # ``as_uri`` percent-encodes the characters that are special in URIs.
171
+ mode = "rw" if write else "ro"
172
+ return f"{path.resolve().as_uri()}?mode={mode}"
173
+
174
+
175
+ def translate_error(error: sqlite3.Error, *, write: bool) -> ExplorerError:
176
+ """Turn a raw sqlite3 error into the ExplorerError shown to the user."""
177
+ message = str(error)
178
+ if not write and "readonly database" in message:
179
+ return ReadOnlyError(message)
180
+ return ExplorerError(message)
181
+
182
+
183
+ @contextmanager
184
+ def open_database(path: Path, *, write: bool = False) -> Iterator[Explorer]:
185
+ """Open the database at *path* and yield an :class:`Explorer` bound to it.
186
+
187
+ The file is opened read-only unless *write* is true, and it is never
188
+ created. In write mode the transaction is committed when the ``with``
189
+ block finishes normally and rolled back if it raises.
190
+ """
191
+ try:
192
+ connection = sqlite3.connect(_database_uri(path, write=write), uri=True)
193
+ except sqlite3.Error as error:
194
+ raise ExplorerError(f"cannot open {path}: {error}") from error
195
+ try:
196
+ yield Explorer(connection, write=write)
197
+ if write:
198
+ connection.commit()
199
+ except sqlite3.Error as error:
200
+ connection.rollback()
201
+ raise translate_error(error, write=write) from error
202
+ finally:
203
+ connection.close()
204
+
205
+
206
+ class Explorer:
207
+ """Explorer-oriented queries on top of an open :class:`sqlite3.Connection`."""
208
+
209
+ def __init__(self, connection: sqlite3.Connection, *, write: bool = False) -> None:
210
+ self._connection = connection
211
+ self._write = write
212
+
213
+ # --- Statements -------------------------------------------------------
214
+
215
+ def execute(
216
+ self, sql: str, parameters: Parameters = (), *, page: Page | None = None
217
+ ) -> ResultSet:
218
+ """Run a single SQL statement and collect its outcome.
219
+
220
+ With *page*, only that window of rows is kept in memory; the rest of
221
+ the result is counted as it streams by and reported as ``total``.
222
+ """
223
+ stream = self.stream(sql, parameters)
224
+ if page is None or not stream.returns_rows:
225
+ return stream.collect()
226
+ return _window(stream, page)
227
+
228
+ def stream(self, sql: str, parameters: Parameters = ()) -> RowStream:
229
+ """Run a single SQL statement and return its rows lazily."""
230
+ cursor = self._connection.execute(sql, parameters)
231
+ if cursor.description is None:
232
+ rowcount = cursor.rowcount
233
+ cursor.close()
234
+ return RowStream(rowcount=rowcount)
235
+ columns = tuple(name for name, *_ in cursor.description)
236
+ return RowStream(columns=columns, rows=_rows_of(cursor), rowcount=cursor.rowcount)
237
+
238
+ def explain(self, sql: str, parameters: Parameters = ()) -> ResultSet:
239
+ """Return the query plan of *sql* as an indented tree."""
240
+ return plan_tree(self.execute(f"EXPLAIN QUERY PLAN {sql}", parameters))
241
+
242
+ def commit(self) -> None:
243
+ self._connection.commit()
244
+
245
+ def rollback(self) -> None:
246
+ self._connection.rollback()
247
+
248
+ def attach(self, alias: str, path: Path, *, write: bool | None = None) -> None:
249
+ """Attach the database at *path* under *alias*, read-only unless *write*."""
250
+ if write is None:
251
+ write = self._write
252
+ try:
253
+ self._connection.execute(
254
+ f"ATTACH DATABASE ? AS {quote_identifier(alias)}",
255
+ (_database_uri(path, write=write),),
256
+ )
257
+ except sqlite3.Error as error:
258
+ raise ExplorerError(f"cannot attach {path}: {error}") from error
259
+
260
+ # --- Catalogue --------------------------------------------------------
261
+
262
+ def names(
263
+ self, *, kinds: Sequence[str] = ("table", "view"), include_internal: bool = False
264
+ ) -> list[str]:
265
+ """Names of the schema objects of the given *kinds*, sorted."""
266
+ placeholders = ", ".join("?" for _ in kinds)
267
+ sql = f"SELECT name FROM sqlite_master WHERE type IN ({placeholders})"
268
+ if not include_internal:
269
+ sql += " AND name NOT LIKE 'sqlite_%'"
270
+ sql += " ORDER BY name"
271
+ return [row[0] for row in self.execute(sql, tuple(kinds)).rows]
272
+
273
+ def objects(self, *, include_internal: bool = False) -> list[tuple[str, str, str]]:
274
+ """``(type, name, sql)`` of every object with a CREATE statement, in creation order."""
275
+ sql = "SELECT type, name, sql FROM sqlite_master WHERE sql IS NOT NULL"
276
+ if not include_internal:
277
+ sql += " AND name NOT LIKE 'sqlite_%'"
278
+ # Creation order, like the sqlite3 shell: every object comes after
279
+ # the ones it depends on, so the output can be replayed as a script.
280
+ sql += " ORDER BY rowid"
281
+ return [(kind, name, statement) for kind, name, statement in self.execute(sql).rows]
282
+
283
+ def tables(self, *, include_internal: bool = False, count: bool = True) -> ResultSet:
284
+ """List tables and views, with their row counts unless *count* is false."""
285
+ sql = "SELECT type, name FROM sqlite_master WHERE type IN ('table', 'view')"
286
+ if not include_internal:
287
+ sql += " AND name NOT LIKE 'sqlite_%'"
288
+ sql += " ORDER BY type, name"
289
+ objects = self.execute(sql).rows
290
+ if not count:
291
+ return ResultSet(columns=("type", "name"), rows=[tuple(row) for row in objects])
292
+ rows = [(kind, name, self._count_rows(name)) for kind, name in objects]
293
+ return ResultSet(columns=("type", "name", "rows"), rows=rows)
294
+
295
+ def _count_rows(self, name: str) -> int | str:
296
+ try:
297
+ return self.execute(f"SELECT COUNT(*) FROM {quote_identifier(name)}").rows[0][0]
298
+ except sqlite3.Error:
299
+ # Typically a view whose underlying table no longer exists.
300
+ return "?"
301
+
302
+ def schema(self, name: str | None = None, *, include_internal: bool = False) -> list[str]:
303
+ """Return the CREATE statements of the database.
304
+
305
+ When *name* is given, only the table or view itself plus its indexes
306
+ and triggers are returned. Otherwise SQLite's internal objects
307
+ (``sqlite_*``) are skipped unless *include_internal* is true.
308
+ """
309
+ if name is None:
310
+ return [
311
+ statement for _, _, statement in self.objects(include_internal=include_internal)
312
+ ]
313
+ result = self.execute(
314
+ "SELECT sql FROM sqlite_master WHERE sql IS NOT NULL"
315
+ " AND (name = ? OR tbl_name = ?) ORDER BY rowid",
316
+ (name, name),
317
+ )
318
+ statements = [row[0] for row in result.rows]
319
+ if not statements:
320
+ raise ExplorerError(f"no such table or view: {name}")
321
+ return statements
322
+
323
+ def columns(self, table: str) -> ResultSet:
324
+ """Describe the columns of *table* as reported by ``PRAGMA table_info``."""
325
+ result = self.execute(f"PRAGMA table_info({quote_identifier(table)})")
326
+ if not result.rows:
327
+ raise ExplorerError(f"no such table or view: {table}")
328
+ return result
329
+
330
+ def column_names(self, table: str) -> list[str]:
331
+ return [row[1] for row in self.columns(table).rows]
332
+
333
+ def _resolve_column(self, table: str, name: str, known: Sequence[str]) -> str:
334
+ for column in known:
335
+ if column.casefold() == name.casefold():
336
+ return column
337
+ raise ExplorerError(f"no such column in {table}: {name}")
338
+
339
+ def indexes(self, table: str | None = None) -> ResultSet:
340
+ """Indexes of *table*, or of every table, with the columns they cover."""
341
+ tables = self._tables_or_one(table)
342
+ rows = []
343
+ for name in tables:
344
+ index_list = self.execute(f"PRAGMA index_list({quote_identifier(name)})").rows
345
+ for _seq, index, unique, origin, partial in index_list:
346
+ info = self.execute(f"PRAGMA index_info({quote_identifier(index)})").rows
347
+ covered = ", ".join("<expr>" if column is None else column for _, _, column in info)
348
+ rows.append((name, index, unique, origin, partial, covered))
349
+ return ResultSet(
350
+ columns=("table", "name", "unique", "origin", "partial", "columns"), rows=rows
351
+ )
352
+
353
+ def foreign_keys(self, table: str | None = None) -> ResultSet:
354
+ """Foreign keys declared by *table*, or by every table."""
355
+ tables = self._tables_or_one(table)
356
+ rows = []
357
+ for name in tables:
358
+ fks = self.execute(f"PRAGMA foreign_key_list({quote_identifier(name)})").rows
359
+ for fk_id, seq, referenced, source, target, on_update, on_delete, match in fks:
360
+ rows.append(
361
+ (name, fk_id, seq, source, referenced, target, on_update, on_delete, match)
362
+ )
363
+ return ResultSet(
364
+ columns=(
365
+ "table",
366
+ "id",
367
+ "seq",
368
+ "from",
369
+ "references",
370
+ "to",
371
+ "on_update",
372
+ "on_delete",
373
+ "match",
374
+ ),
375
+ rows=rows,
376
+ )
377
+
378
+ def _tables_or_one(self, table: str | None) -> list[str]:
379
+ if table is None:
380
+ return self.names(kinds=("table",))
381
+ self.columns(table) # fail early with a clear message
382
+ return [table]
383
+
384
+ def info(self, *, check: bool = False) -> ResultSet:
385
+ """Facts about the database file: size, pragmas and object counts."""
386
+ databases = self.execute("PRAGMA database_list").rows
387
+ path = next((file for _, name, file in databases if name == "main"), "")
388
+ try:
389
+ size: int | None = Path(path).stat().st_size if path else None
390
+ except OSError:
391
+ size = None
392
+ entries: list[tuple[str, object]] = [
393
+ ("path", path),
394
+ ("size", size),
395
+ ("sqlite_version", sqlite3.sqlite_version),
396
+ ]
397
+ pragmas = (
398
+ "page_size",
399
+ "page_count",
400
+ "freelist_count",
401
+ "journal_mode",
402
+ "encoding",
403
+ "user_version",
404
+ "application_id",
405
+ )
406
+ for pragma in pragmas:
407
+ entries.append((pragma, self.execute(f"PRAGMA {pragma}").rows[0][0]))
408
+ counts = dict(
409
+ self.execute(
410
+ "SELECT type, COUNT(*) FROM sqlite_master"
411
+ " WHERE name NOT LIKE 'sqlite_%' GROUP BY type"
412
+ ).rows
413
+ )
414
+ plurals = {"table": "tables", "view": "views", "index": "indexes", "trigger": "triggers"}
415
+ for kind, plural in plurals.items():
416
+ entries.append((plural, counts.get(kind, 0)))
417
+ if check:
418
+ problems = self.execute("PRAGMA integrity_check").rows
419
+ entries.append(("integrity_check", "; ".join(str(row[0]) for row in problems)))
420
+ return ResultSet(columns=("key", "value"), rows=entries)
421
+
422
+ # --- Data -------------------------------------------------------------
423
+
424
+ def _rows_sql(
425
+ self,
426
+ table: str,
427
+ *,
428
+ columns: Sequence[str] | None,
429
+ where: str | None,
430
+ order_by: str | None,
431
+ descending: bool,
432
+ ) -> tuple[str, str, str]:
433
+ """``(selection, source, order)`` pieces of a SELECT over *table*."""
434
+ selection = "*"
435
+ order_clause = ""
436
+ if columns or order_by:
437
+ known = self.column_names(table)
438
+ if columns:
439
+ selection = ", ".join(
440
+ quote_identifier(self._resolve_column(table, name, known)) for name in columns
441
+ )
442
+ if order_by:
443
+ column = quote_identifier(self._resolve_column(table, order_by, known))
444
+ order_clause = f" ORDER BY {column}{' DESC' if descending else ''}"
445
+ source = f"FROM {quote_identifier(table)}"
446
+ if where:
447
+ source += f" WHERE ({where})"
448
+ return selection, source, order_clause
449
+
450
+ def rows(
451
+ self,
452
+ table: str,
453
+ *,
454
+ columns: Sequence[str] | None = None,
455
+ where: str | None = None,
456
+ order_by: str | None = None,
457
+ descending: bool = False,
458
+ limit: int | None = None,
459
+ offset: int = 0,
460
+ page: Page | None = None,
461
+ ) -> ResultSet:
462
+ """Return the rows of *table*, optionally filtered, ordered and windowed.
463
+
464
+ *where* is a raw SQL fragment; *columns* and *order_by* are validated
465
+ against the table's columns (case-insensitively). *limit* and *offset*
466
+ select the rows; *page* then picks one page of them in SQL, so only
467
+ that page is ever loaded, and ``total`` reports how many there were.
468
+ """
469
+ selection, source, order = self._rows_sql(
470
+ table, columns=columns, where=where, order_by=order_by, descending=descending
471
+ )
472
+ query = f"SELECT {selection} {source}{order} LIMIT ? OFFSET ?"
473
+ if page is None:
474
+ return self.execute(query, (-1 if limit is None else limit, offset))
475
+ matching = self.execute(f"SELECT COUNT(*) {source}").rows[0][0]
476
+ total = max(0, matching - offset)
477
+ if limit is not None:
478
+ total = min(total, limit)
479
+ size = max(0, min(page.size, total - page.start))
480
+ window = self.execute(query, (size, offset + page.start))
481
+ return ResultSet(
482
+ columns=window.columns, rows=window.rows, rowcount=window.rowcount, total=total
483
+ )
484
+
485
+ def stream_rows(
486
+ self,
487
+ table: str,
488
+ *,
489
+ columns: Sequence[str] | None = None,
490
+ where: str | None = None,
491
+ order_by: str | None = None,
492
+ descending: bool = False,
493
+ limit: int | None = None,
494
+ offset: int = 0,
495
+ ) -> RowStream:
496
+ """Like :meth:`rows`, but the rows come lazily from the cursor."""
497
+ selection, source, order = self._rows_sql(
498
+ table, columns=columns, where=where, order_by=order_by, descending=descending
499
+ )
500
+ query = f"SELECT {selection} {source}{order} LIMIT ? OFFSET ?"
501
+ return self.stream(query, (-1 if limit is None else limit, offset))
502
+
503
+ def stats(
504
+ self,
505
+ table: str,
506
+ *,
507
+ top: int = 3,
508
+ columns: Sequence[str] | None = None,
509
+ sample: int | None = None,
510
+ ) -> StatsReport:
511
+ """Per-column summary of *table*: nulls, distinct values, min, max, top values.
512
+
513
+ Nulls, distinct values, minimum and maximum of every column come from a
514
+ single pass over the table; the *top* most frequent values need one
515
+ ``GROUP BY`` per column, so ``top=0`` is much cheaper on big tables.
516
+ With *sample*, everything is computed on a random sample of that many
517
+ rows copied to a temporary table.
518
+ """
519
+ info = self.columns(table).rows
520
+ known = [row[1] for row in info]
521
+ declared = {row[1]: row[2] for row in info}
522
+ if columns is None:
523
+ selected = known
524
+ else:
525
+ selected = [self._resolve_column(table, name, known) for name in columns]
526
+ if sample is None:
527
+ return self._collect_stats(quote_identifier(table), selected, declared, top, False)
528
+ source = self._sample(table, selected, sample)
529
+ try:
530
+ return self._collect_stats(source, selected, declared, top, True)
531
+ finally:
532
+ self._connection.execute(f"DROP TABLE IF EXISTS {source}")
533
+
534
+ def _sample(self, table: str, columns: Sequence[str], size: int) -> str:
535
+ """Copy a random sample of *table* into a temporary table; return its name."""
536
+ selection = ", ".join(quote_identifier(name) for name in columns)
537
+ self._connection.execute(f"DROP TABLE IF EXISTS {_SAMPLE_TABLE}")
538
+ self._connection.execute(
539
+ f"CREATE TEMP TABLE {_SAMPLE_NAME} AS SELECT {selection}"
540
+ f" FROM {quote_identifier(table)} ORDER BY random() LIMIT ?",
541
+ (size,),
542
+ )
543
+ return _SAMPLE_TABLE
544
+
545
+ def _collect_stats(
546
+ self,
547
+ source: str,
548
+ columns: Sequence[str],
549
+ declared: Mapping[str, str],
550
+ top: int,
551
+ sampled: bool,
552
+ ) -> StatsReport:
553
+ rows = []
554
+ examined = 0
555
+ for start in range(0, len(columns), _STATS_CHUNK):
556
+ chunk = columns[start : start + _STATS_CHUNK]
557
+ expressions = ["COUNT(*)"]
558
+ for name in chunk:
559
+ q_col = quote_identifier(name)
560
+ expressions += [
561
+ f"COUNT({q_col})",
562
+ f"COUNT(DISTINCT {q_col})",
563
+ f"MIN({q_col})",
564
+ f"MAX({q_col})",
565
+ f"typeof(MIN({q_col}))",
566
+ ]
567
+ values = self.execute(f"SELECT {', '.join(expressions)} FROM {source}").rows[0]
568
+ examined = values[0]
569
+ for index, name in enumerate(chunk):
570
+ filled, distinct, minimum, maximum, min_type = values[1 + index * 5 : 6 + index * 5]
571
+ rows.append(
572
+ (
573
+ name,
574
+ declared.get(name) or min_type,
575
+ examined - filled,
576
+ distinct,
577
+ minimum,
578
+ maximum,
579
+ self._top_values(source, name, top),
580
+ )
581
+ )
582
+ if not columns:
583
+ examined = self.execute(f"SELECT COUNT(*) FROM {source}").rows[0][0]
584
+ return StatsReport(ResultSet(columns=_STATS_COLUMNS, rows=rows), examined, sampled)
585
+
586
+ def _top_values(self, source: str, column: str, top: int) -> str:
587
+ if top <= 0:
588
+ return ""
589
+ q_col = quote_identifier(column)
590
+ frequent = self.execute(
591
+ f"SELECT quote({q_col}), COUNT(*) AS n FROM {source}"
592
+ f" WHERE {q_col} IS NOT NULL GROUP BY {q_col} ORDER BY n DESC, 1 LIMIT ?",
593
+ (top,),
594
+ ).rows
595
+ return ", ".join(f"{value} ({count})" for value, count in frequent)
596
+
597
+ def search(
598
+ self, text: str, *, tables: Sequence[str] | None = None, limit: int | None = None
599
+ ) -> ResultSet:
600
+ """Find *text* (case-insensitive substring) in every non-BLOB column.
601
+
602
+ Each table is scanned once, whatever its number of columns; one row is
603
+ reported per matching column.
604
+ """
605
+ if not text:
606
+ raise ExplorerError("nothing to search for")
607
+ escaped = text.replace("\\", "\\\\").replace("%", "\\%").replace("_", "\\_")
608
+ pattern = f"%{escaped}%"
609
+ if tables is None:
610
+ tables = self.names(kinds=("table",))
611
+ remaining = limit
612
+ rows = []
613
+ for table in tables:
614
+ if remaining is not None and remaining <= 0:
615
+ break
616
+ columns = self.column_names(table)
617
+ for record in self._scan_table(table, columns, pattern, remaining):
618
+ rowid, values, matched = (
619
+ record[0],
620
+ record[1:][: len(columns)],
621
+ record[1 + len(columns) :],
622
+ )
623
+ for column, value, hit in zip(columns, values, matched, strict=True):
624
+ if not hit:
625
+ continue
626
+ rows.append((table, column, rowid, value))
627
+ if remaining is not None:
628
+ remaining -= 1
629
+ if remaining <= 0:
630
+ break
631
+ if remaining is not None and remaining <= 0:
632
+ break
633
+ return ResultSet(columns=_SEARCH_COLUMNS, rows=rows)
634
+
635
+ def _scan_table(
636
+ self, table: str, columns: Sequence[str], pattern: str, limit: int | None
637
+ ) -> list[tuple]:
638
+ """One pass over *table* returning ``rowid, values..., matched flags...`` per hit row."""
639
+ quoted = [quote_identifier(column) for column in columns]
640
+ flags = [f'"__match_{index}"' for index in range(len(columns))]
641
+ tests = ", ".join(
642
+ f"(typeof({column}) <> 'blob' AND CAST({column} AS TEXT) LIKE ? ESCAPE '\\') AS {flag}"
643
+ for column, flag in zip(quoted, flags, strict=True)
644
+ )
645
+ selection = ", ".join(quoted)
646
+ parameters = (*[pattern] * len(columns), -1 if limit is None else limit)
647
+
648
+ def sql(rowid: str) -> str:
649
+ return (
650
+ f'SELECT * FROM (SELECT {rowid} AS "__rowid", {selection}, {tests}'
651
+ f" FROM {quote_identifier(table)}) WHERE {' OR '.join(flags)} LIMIT ?"
652
+ )
653
+
654
+ try:
655
+ return self.execute(sql("rowid"), parameters).rows
656
+ except sqlite3.OperationalError as error:
657
+ if "no such column: rowid" not in str(error):
658
+ raise
659
+ # WITHOUT ROWID tables and views have no rowid to report.
660
+ return self.execute(sql("NULL"), parameters).rows
661
+
662
+ def dump(self) -> Iterator[str]:
663
+ """The database as replayable SQL, one statement per item."""
664
+ return self._connection.iterdump()
665
+
666
+ def import_rows(
667
+ self,
668
+ table: str,
669
+ columns: Sequence[tuple[str, str]],
670
+ rows: Iterable[Sequence[object]],
671
+ *,
672
+ create: bool = True,
673
+ ) -> int:
674
+ """Insert *rows* into *table*, creating it from *columns* (``(name, type)``) if needed."""
675
+ q_table = quote_identifier(table)
676
+ exists = self.execute(
677
+ "SELECT 1 FROM sqlite_master WHERE type = 'table' AND name = ?", (table,)
678
+ ).rows
679
+ if not exists:
680
+ if not create:
681
+ raise ExplorerError(f"no such table: {table}")
682
+ definition = ", ".join(f"{quote_identifier(name)} {kind}" for name, kind in columns)
683
+ self._connection.execute(f"CREATE TABLE {q_table} ({definition})")
684
+ names = ", ".join(quote_identifier(name) for name, _ in columns)
685
+ placeholders = ", ".join("?" for _ in columns)
686
+ cursor = self._connection.executemany(
687
+ f"INSERT INTO {q_table} ({names}) VALUES ({placeholders})", rows
688
+ )
689
+ try:
690
+ return cursor.rowcount
691
+ finally:
692
+ cursor.close()
693
+
694
+
695
+ _SEARCH_COLUMNS = ("table", "column", "rowid", "value")
696
+ _STATS_COLUMNS = ("column", "type", "nulls", "distinct", "min", "max", "top")
697
+ # Columns per aggregate query: five expressions each, well below SQLITE_MAX_COLUMN.
698
+ _STATS_CHUNK = 100
699
+ _SAMPLE_NAME = '"sqlitexplorer_sample"'
700
+ _SAMPLE_TABLE = f"temp.{_SAMPLE_NAME}"