sqlitexplorer 1.0.0__tar.gz → 1.1.2__tar.gz

This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
Files changed (28) hide show
  1. sqlitexplorer-1.1.2/MANIFEST.in +2 -0
  2. {sqlitexplorer-1.0.0/sqlitexplorer.egg-info → sqlitexplorer-1.1.2}/PKG-INFO +22 -24
  3. {sqlitexplorer-1.0.0 → sqlitexplorer-1.1.2}/README.md +20 -23
  4. {sqlitexplorer-1.0.0 → sqlitexplorer-1.1.2}/pyproject.toml +2 -1
  5. {sqlitexplorer-1.0.0 → sqlitexplorer-1.1.2}/sqlitexplorer/charts.py +90 -13
  6. {sqlitexplorer-1.0.0 → sqlitexplorer-1.1.2}/sqlitexplorer/cli.py +75 -15
  7. {sqlitexplorer-1.0.0 → sqlitexplorer-1.1.2}/sqlitexplorer/core.py +4 -1
  8. {sqlitexplorer-1.0.0 → sqlitexplorer-1.1.2}/sqlitexplorer/render.py +28 -9
  9. {sqlitexplorer-1.0.0 → sqlitexplorer-1.1.2/sqlitexplorer.egg-info}/PKG-INFO +22 -24
  10. {sqlitexplorer-1.0.0 → sqlitexplorer-1.1.2}/sqlitexplorer.egg-info/SOURCES.txt +2 -0
  11. {sqlitexplorer-1.0.0 → sqlitexplorer-1.1.2}/sqlitexplorer.egg-info/requires.txt +1 -0
  12. sqlitexplorer-1.1.2/tests/conftest.py +83 -0
  13. sqlitexplorer-1.1.2/tests/test_charts.py +229 -0
  14. {sqlitexplorer-1.0.0 → sqlitexplorer-1.1.2}/tests/test_cli.py +124 -1
  15. {sqlitexplorer-1.0.0 → sqlitexplorer-1.1.2}/tests/test_core.py +12 -0
  16. {sqlitexplorer-1.0.0 → sqlitexplorer-1.1.2}/tests/test_render.py +28 -0
  17. sqlitexplorer-1.0.0/tests/test_charts.py +0 -108
  18. {sqlitexplorer-1.0.0 → sqlitexplorer-1.1.2}/LICENSE +0 -0
  19. {sqlitexplorer-1.0.0 → sqlitexplorer-1.1.2}/setup.cfg +0 -0
  20. {sqlitexplorer-1.0.0 → sqlitexplorer-1.1.2}/sqlitexplorer/__init__.py +0 -0
  21. {sqlitexplorer-1.0.0 → sqlitexplorer-1.1.2}/sqlitexplorer/__main__.py +0 -0
  22. {sqlitexplorer-1.0.0 → sqlitexplorer-1.1.2}/sqlitexplorer/completion.py +0 -0
  23. {sqlitexplorer-1.0.0 → sqlitexplorer-1.1.2}/sqlitexplorer/shell.py +0 -0
  24. {sqlitexplorer-1.0.0 → sqlitexplorer-1.1.2}/sqlitexplorer.egg-info/dependency_links.txt +0 -0
  25. {sqlitexplorer-1.0.0 → sqlitexplorer-1.1.2}/sqlitexplorer.egg-info/entry_points.txt +0 -0
  26. {sqlitexplorer-1.0.0 → sqlitexplorer-1.1.2}/sqlitexplorer.egg-info/top_level.txt +0 -0
  27. {sqlitexplorer-1.0.0 → sqlitexplorer-1.1.2}/tests/test_completion.py +0 -0
  28. {sqlitexplorer-1.0.0 → sqlitexplorer-1.1.2}/tests/test_shell.py +0 -0
@@ -0,0 +1,2 @@
1
+ graft tests
2
+ global-exclude __pycache__ *.py[cod]
@@ -1,6 +1,6 @@
1
1
  Metadata-Version: 2.4
2
2
  Name: sqlitexplorer
3
- Version: 1.0.0
3
+ Version: 1.1.2
4
4
  Summary: Explore SQLite databases from the terminal: tables, schema, stats, search, queries, charts, export/import and an interactive shell.
5
5
  Author-email: "Carlos A. Planchón" <carlosandresplanchonprestes@gmail.com>
6
6
  License-Expression: MIT
@@ -21,6 +21,7 @@ Description-Content-Type: text/markdown
21
21
  License-File: LICENSE
22
22
  Requires-Dist: outfancy>=0.11
23
23
  Requires-Dist: plotille>=6
24
+ Requires-Dist: plotilleresample>=1.0
24
25
  Requires-Dist: typer>=0.12
25
26
  Provides-Extra: dev
26
27
  Requires-Dist: pytest>=8; extra == "dev"
@@ -35,7 +36,8 @@ A command-line explorer for SQLite databases. It lists tables, prints schemas,
35
36
  dumps rows, computes statistics, searches values, runs ad-hoc queries, draws
36
37
  charts, exports and imports data, and offers an interactive shell. Tables are
37
38
  rendered with [outfancy](https://github.com/carlosplanchon/outfancy) and charts
38
- with [plotille](https://github.com/tammoippen/plotille).
39
+ with [plotille](https://github.com/tammoippen/plotille), resampled by
40
+ [plotilleresample](https://github.com/carlosplanchon/plotilleresample).
39
41
 
40
42
  [![CI](https://github.com/carlosplanchon/sqlitexplorer/actions/workflows/ci.yml/badge.svg)](https://github.com/carlosplanchon/sqlitexplorer/actions/workflows/ci.yml)
41
43
  [![PyPI version](https://img.shields.io/pypi/v/sqlitexplorer.svg)](https://pypi.org/project/sqlitexplorer/)
@@ -127,6 +129,8 @@ Big results do not need to fit in memory: `--page` fetches only the requested
127
129
  page (in SQL for `show`, from the cursor for `query`), and the csv, tsv, json
128
130
  and markdown formats, `export` and `dump` are written row by row. The table
129
131
  format is the exception, since it needs every row to size its columns.
132
+ `chart` reduces as it reads, so the memory it needs does not grow with the
133
+ table; `chart --no-resample` is the one that holds every row.
130
134
 
131
135
  ## Queries
132
136
 
@@ -159,6 +163,14 @@ stderr. `--kind` selects `line` (default), `scatter` or `hist` (first column
159
163
  only, `--bins`). `--height`, `--width`, `--x-label`, `--y-label` and `--color`
160
164
  adjust the drawing.
161
165
 
166
+ Line and scatter charts are reduced to what the canvas can show, keeping the
167
+ minimum and the maximum of every column of braille dots, so spikes survive and
168
+ the true extremes keep the X of their own row. The rows are reduced as they are
169
+ read, in chunks, so a chart over millions of rows needs no more memory than one
170
+ over a thousand; the reduction is reported on stderr. `--no-resample` reads and
171
+ plots every row instead. Histograms are never reduced, since dropping rows would
172
+ change the distribution.
173
+
162
174
  ## Export, import, dump and diff
163
175
 
164
176
  ```sh
@@ -170,8 +182,14 @@ sqlitexplorer diff app.db backup.db # exit status 1 when th
170
182
  ```
171
183
 
172
184
  `import` creates the table when it does not exist, inferring INTEGER, REAL or
173
- TEXT for each column; empty CSV cells become NULL. The format comes from the
174
- file extension unless `--format` is given.
185
+ TEXT for each column; values with leading zeros and integers too large for
186
+ SQLite stay TEXT, so codes such as `007` keep their digits. Empty CSV cells
187
+ become NULL. The format comes from the file extension unless `--format` is
188
+ given, and `--encoding` reads a file that is not UTF-8.
189
+
190
+ `export --all` writes one file per table and view inside the directory: a name
191
+ that would not be a valid file name is sanitised, and an object that cannot be
192
+ read (a view over a dropped table) is skipped with a warning on stderr.
175
193
 
176
194
  ## Shell
177
195
 
@@ -204,26 +222,6 @@ uv run pytest
204
222
  uv run ruff check .
205
223
  ```
206
224
 
207
- ## Releasing
208
-
209
- Releases are driven by version tags. Pushing `vX.Y.Z` runs the tests, builds
210
- the distributions, publishes them to PyPI with
211
- [trusted publishing](https://docs.pypi.org/trusted-publishers/) and creates a
212
- GitHub release with the artifacts attached.
213
-
214
- ```sh
215
- uv version 0.3.0 # or: uv version --bump minor
216
- git commit -am "Release 0.3.0"
217
- git tag v0.3.0
218
- git push origin master v0.3.0
219
- ```
220
-
221
- The tag must match the version in `pyproject.toml`; the workflow refuses to
222
- publish otherwise. Before the first release, register the repository as a
223
- trusted publisher of the project on PyPI with the workflow name `release.yml`
224
- and the environment `pypi`, and create that environment in the repository
225
- settings on GitHub.
226
-
227
225
  ## License
228
226
 
229
227
  MIT. See [LICENSE](LICENSE).
@@ -6,7 +6,8 @@ A command-line explorer for SQLite databases. It lists tables, prints schemas,
6
6
  dumps rows, computes statistics, searches values, runs ad-hoc queries, draws
7
7
  charts, exports and imports data, and offers an interactive shell. Tables are
8
8
  rendered with [outfancy](https://github.com/carlosplanchon/outfancy) and charts
9
- with [plotille](https://github.com/tammoippen/plotille).
9
+ with [plotille](https://github.com/tammoippen/plotille), resampled by
10
+ [plotilleresample](https://github.com/carlosplanchon/plotilleresample).
10
11
 
11
12
  [![CI](https://github.com/carlosplanchon/sqlitexplorer/actions/workflows/ci.yml/badge.svg)](https://github.com/carlosplanchon/sqlitexplorer/actions/workflows/ci.yml)
12
13
  [![PyPI version](https://img.shields.io/pypi/v/sqlitexplorer.svg)](https://pypi.org/project/sqlitexplorer/)
@@ -98,6 +99,8 @@ Big results do not need to fit in memory: `--page` fetches only the requested
98
99
  page (in SQL for `show`, from the cursor for `query`), and the csv, tsv, json
99
100
  and markdown formats, `export` and `dump` are written row by row. The table
100
101
  format is the exception, since it needs every row to size its columns.
102
+ `chart` reduces as it reads, so the memory it needs does not grow with the
103
+ table; `chart --no-resample` is the one that holds every row.
101
104
 
102
105
  ## Queries
103
106
 
@@ -130,6 +133,14 @@ stderr. `--kind` selects `line` (default), `scatter` or `hist` (first column
130
133
  only, `--bins`). `--height`, `--width`, `--x-label`, `--y-label` and `--color`
131
134
  adjust the drawing.
132
135
 
136
+ Line and scatter charts are reduced to what the canvas can show, keeping the
137
+ minimum and the maximum of every column of braille dots, so spikes survive and
138
+ the true extremes keep the X of their own row. The rows are reduced as they are
139
+ read, in chunks, so a chart over millions of rows needs no more memory than one
140
+ over a thousand; the reduction is reported on stderr. `--no-resample` reads and
141
+ plots every row instead. Histograms are never reduced, since dropping rows would
142
+ change the distribution.
143
+
133
144
  ## Export, import, dump and diff
134
145
 
135
146
  ```sh
@@ -141,8 +152,14 @@ sqlitexplorer diff app.db backup.db # exit status 1 when th
141
152
  ```
142
153
 
143
154
  `import` creates the table when it does not exist, inferring INTEGER, REAL or
144
- TEXT for each column; empty CSV cells become NULL. The format comes from the
145
- file extension unless `--format` is given.
155
+ TEXT for each column; values with leading zeros and integers too large for
156
+ SQLite stay TEXT, so codes such as `007` keep their digits. Empty CSV cells
157
+ become NULL. The format comes from the file extension unless `--format` is
158
+ given, and `--encoding` reads a file that is not UTF-8.
159
+
160
+ `export --all` writes one file per table and view inside the directory: a name
161
+ that would not be a valid file name is sanitised, and an object that cannot be
162
+ read (a view over a dropped table) is skipped with a warning on stderr.
146
163
 
147
164
  ## Shell
148
165
 
@@ -175,26 +192,6 @@ uv run pytest
175
192
  uv run ruff check .
176
193
  ```
177
194
 
178
- ## Releasing
179
-
180
- Releases are driven by version tags. Pushing `vX.Y.Z` runs the tests, builds
181
- the distributions, publishes them to PyPI with
182
- [trusted publishing](https://docs.pypi.org/trusted-publishers/) and creates a
183
- GitHub release with the artifacts attached.
184
-
185
- ```sh
186
- uv version 0.3.0 # or: uv version --bump minor
187
- git commit -am "Release 0.3.0"
188
- git tag v0.3.0
189
- git push origin master v0.3.0
190
- ```
191
-
192
- The tag must match the version in `pyproject.toml`; the workflow refuses to
193
- publish otherwise. Before the first release, register the repository as a
194
- trusted publisher of the project on PyPI with the workflow name `release.yml`
195
- and the environment `pypi`, and create that environment in the repository
196
- settings on GitHub.
197
-
198
195
  ## License
199
196
 
200
197
  MIT. See [LICENSE](LICENSE).
@@ -4,7 +4,7 @@ build-backend = "setuptools.build_meta"
4
4
 
5
5
  [project]
6
6
  name = "sqlitexplorer"
7
- version = "1.0.0"
7
+ version = "1.1.2"
8
8
  description = "Explore SQLite databases from the terminal: tables, schema, stats, search, queries, charts, export/import and an interactive shell."
9
9
  readme = "README.md"
10
10
  license = "MIT"
@@ -29,6 +29,7 @@ classifiers = [
29
29
  dependencies = [
30
30
  "outfancy>=0.11",
31
31
  "plotille>=6",
32
+ "plotilleresample>=1.0",
32
33
  "typer>=0.12",
33
34
  ]
34
35
 
@@ -12,10 +12,12 @@ from contextlib import contextmanager
12
12
  from dataclasses import dataclass
13
13
  from datetime import datetime
14
14
  from enum import Enum
15
+ from itertools import islice
15
16
 
16
17
  import plotille
18
+ import plotilleresample
17
19
 
18
- from sqlitexplorer.core import ExplorerError, ResultSet
20
+ from sqlitexplorer.core import ExplorerError, ResultSet, RowStream
19
21
 
20
22
  __all__ = [
21
23
  "ChartKind",
@@ -23,7 +25,9 @@ __all__ = [
23
25
  "histogram_values",
24
26
  "render_chart",
25
27
  "render_histogram",
28
+ "resample_series",
26
29
  "series_from_result",
30
+ "stream_series",
27
31
  ]
28
32
 
29
33
  PALETTE = ("red", "green", "yellow", "blue", "magenta", "cyan")
@@ -31,6 +35,13 @@ PALETTE = ("red", "green", "yellow", "blue", "magenta", "cyan")
31
35
  AXIS_LABEL_WIDTH = 8
32
36
  # Characters taken by the Y axis (ticks, label and separator) next to the canvas.
33
37
  AXIS_WIDTH = 12
38
+ # A tuple, not bool | int | float: the union would be rebuilt on every call,
39
+ # and this runs once per value of the result.
40
+ _NUMERIC = (bool, int, float)
41
+ # Rows read at a time when streaming, and how many reduced points may pile
42
+ # up before they are reduced again.
43
+ _CHUNK = 65536
44
+ _PILE = 8
34
45
 
35
46
 
36
47
  class ChartKind(str, Enum):
@@ -47,7 +58,7 @@ class Series:
47
58
 
48
59
 
49
60
  def _number(value: object) -> float | None:
50
- if isinstance(value, bool | int | float):
61
+ if isinstance(value, _NUMERIC):
51
62
  return float(value)
52
63
  if isinstance(value, str):
53
64
  try:
@@ -69,11 +80,13 @@ def _x_value(value: object) -> float | datetime | None:
69
80
  return None
70
81
 
71
82
 
72
- def series_from_result(result: ResultSet) -> tuple[list[Series], int]:
83
+ def series_from_result(result: ResultSet, *, require_rows: bool = True) -> tuple[list[Series], int]:
73
84
  """Split *result* into one series per numeric column after the first.
74
85
 
75
86
  Rows with a NULL in the X column or in any series are skipped; the second
76
- item of the returned tuple counts them.
87
+ item of the returned tuple counts them. With *require_rows* false an empty
88
+ result yields empty series instead of raising, which is what the streaming
89
+ reader needs for a chunk that holds nothing usable.
77
90
  """
78
91
  if len(result.columns) < 2:
79
92
  raise ExplorerError("need an X column and at least one numeric column")
@@ -82,8 +95,10 @@ def series_from_result(result: ResultSet) -> tuple[list[Series], int]:
82
95
  ys: list[list[float]] = [[] for _ in y_names]
83
96
  x_type: type | None = None
84
97
  skipped = 0
98
+ keep_x = xs.append
99
+ keep = [bucket.append for bucket in ys]
85
100
  for row in result.rows:
86
- if any(value is None for value in row):
101
+ if None in row:
87
102
  skipped += 1
88
103
  continue
89
104
  x = _x_value(row[0])
@@ -93,26 +108,88 @@ def series_from_result(result: ResultSet) -> tuple[list[Series], int]:
93
108
  x_type = type(x)
94
109
  elif not isinstance(x, x_type):
95
110
  raise ExplorerError(f"column {x_name} mixes numbers and dates")
96
- numbers = []
97
- for name, value in zip(y_names, row[1:], strict=True):
111
+ for name, value, append in zip(y_names, row[1:], keep, strict=True):
98
112
  number = _number(value)
99
113
  if number is None:
100
114
  raise ExplorerError(f"column {name} is not numeric: {value!r}")
101
- numbers.append(number)
102
- xs.append(x)
103
- for bucket, number in zip(ys, numbers, strict=True):
104
- bucket.append(number)
105
- if not xs:
115
+ append(number)
116
+ keep_x(x)
117
+ if require_rows and not xs:
106
118
  raise ExplorerError("no rows to plot")
107
119
  return [
108
120
  Series(label=name, x=xs, y=bucket) for name, bucket in zip(y_names, ys, strict=True)
109
121
  ], skipped
110
122
 
111
123
 
124
+ def resample_series(
125
+ series: Sequence[Series], *, kind: ChartKind, width: int, height: int
126
+ ) -> list[Series]:
127
+ """Reduce every series to the points the canvas can actually draw.
128
+
129
+ min/max keeps the extremes of every bucket, so spikes survive, and it only
130
+ indexes X, which the LTTB resamplers cannot do when X is a date.
131
+ """
132
+ budget = _canvas_width(width)
133
+ reduce = (
134
+ plotilleresample.resample_scatter
135
+ if kind is ChartKind.SCATTER
136
+ else plotilleresample.resample_plot_minmax
137
+ )
138
+ reduced = []
139
+ for item in series:
140
+ x, y = reduce(item.x, item.y, budget, height)
141
+ reduced.append(Series(label=item.label, x=list(x), y=list(y)))
142
+ return reduced
143
+
144
+
145
+ def stream_series(
146
+ stream: RowStream, *, kind: ChartKind, width: int, height: int
147
+ ) -> tuple[list[Series], int, int]:
148
+ """Reduce *stream* to the canvas without ever holding every row.
149
+
150
+ Rows are read in chunks, each chunk is reduced on its own and the reduced
151
+ points are reduced again as they pile up, so the memory a chart needs stops
152
+ growing with the size of the table. Returns the series, how many rows were
153
+ skipped for their NULLs and how many were read.
154
+ """
155
+ reduced: list[Series] = []
156
+ x_type: type | None = None
157
+ skipped = rows_read = 0
158
+ while True:
159
+ rows = list(islice(stream.rows, _CHUNK))
160
+ if not rows:
161
+ break
162
+ rows_read += len(rows)
163
+ chunk, chunk_skipped = series_from_result(
164
+ ResultSet(columns=stream.columns, rows=rows), require_rows=False
165
+ )
166
+ skipped += chunk_skipped
167
+ if not chunk[0].x:
168
+ continue
169
+ if x_type is None:
170
+ x_type = type(chunk[0].x[0])
171
+ elif not isinstance(chunk[0].x[0], x_type):
172
+ raise ExplorerError(f"column {stream.columns[0]} mixes numbers and dates")
173
+ chunk = resample_series(chunk, kind=kind, width=width, height=height)
174
+ reduced = (
175
+ [
176
+ Series(label=old.label, x=old.x + new.x, y=old.y + new.y)
177
+ for old, new in zip(reduced, chunk, strict=True)
178
+ ]
179
+ if reduced
180
+ else chunk
181
+ )
182
+ if len(reduced[0].x) > _PILE * len(chunk[0].x):
183
+ reduced = resample_series(reduced, kind=kind, width=width, height=height)
184
+ if not reduced or not reduced[0].x:
185
+ raise ExplorerError("no rows to plot")
186
+ return resample_series(reduced, kind=kind, width=width, height=height), skipped, rows_read
187
+
188
+
112
189
  def histogram_values(result: ResultSet) -> tuple[list[float], int]:
113
190
  """Numeric values of the first column of *result*, and how many NULLs were skipped."""
114
191
  if not result.columns:
115
- raise ExplorerError("no rows to plot")
192
+ raise ExplorerError("need a numeric column to plot")
116
193
  name = result.columns[0]
117
194
  values: list[float] = []
118
195
  skipped = 0
@@ -11,6 +11,7 @@ from __future__ import annotations
11
11
  import difflib
12
12
  import functools
13
13
  import inspect
14
+ import re
14
15
  import shutil
15
16
  import sys
16
17
  import time
@@ -28,6 +29,7 @@ from sqlitexplorer.charts import (
28
29
  render_chart,
29
30
  render_histogram,
30
31
  series_from_result,
32
+ stream_series,
31
33
  )
32
34
  from sqlitexplorer.completion import complete_table
33
35
  from sqlitexplorer.core import (
@@ -156,6 +158,7 @@ _EXTENSIONS = {
156
158
  OutputFormat.MARKDOWN: "md",
157
159
  }
158
160
  _FORMAT_BY_SUFFIX = {".csv": OutputFormat.CSV, ".tsv": OutputFormat.TSV, ".json": OutputFormat.JSON}
161
+ _UNSAFE_IN_NAME = re.compile(r"[^\w.-]")
159
162
 
160
163
 
161
164
  # --- Helpers ------------------------------------------------------------------
@@ -273,11 +276,31 @@ def _attach_all(db: Explorer, values: Sequence[str] | None, *, write: bool) -> N
273
276
  db.attach(alias, path, write=write)
274
277
 
275
278
 
279
+ def _read_text(file: Path, encoding: str) -> str:
280
+ """Read *file* as text, reporting a wrong encoding as a plain message."""
281
+ try:
282
+ return file.read_text(encoding=encoding)
283
+ except UnicodeDecodeError:
284
+ _fail(f"{file.name} is not valid {encoding} text")
285
+ except LookupError:
286
+ _fail(f"unknown encoding: {encoding}")
287
+ except OSError as error:
288
+ _fail(str(error))
289
+
290
+
291
+ def _export_file_name(name: str, extension: str) -> str:
292
+ """A file name for *name* that cannot escape the directory it is written in."""
293
+ safe = _UNSAFE_IN_NAME.sub("_", name).lstrip(".")
294
+ return f"{safe or '_'}.{extension}"
295
+
296
+
276
297
  def _read_sql(sql: str | None, file: Path | None) -> list[str]:
277
- if (sql is None) == (file is None):
298
+ if sql is not None and file is not None:
278
299
  _fail("give either an SQL statement or --file, not both")
300
+ if sql is None and file is None:
301
+ _fail("give an SQL statement, - to read it from stdin, or --file")
279
302
  if file is not None:
280
- text = file.read_text(encoding="utf-8")
303
+ text = _read_text(file, "utf-8")
281
304
  elif sql == "-":
282
305
  text = sys.stdin.read()
283
306
  else:
@@ -617,7 +640,8 @@ def chart(
617
640
  str,
618
641
  typer.Argument(
619
642
  show_default=False,
620
- help="Query whose first column is X and the other numeric columns are series.",
643
+ help="Query whose first column is X and the other numeric columns are series,"
644
+ " or - to read it from stdin.",
621
645
  ),
622
646
  ],
623
647
  kind: Annotated[
@@ -625,6 +649,13 @@ def chart(
625
649
  ] = ChartKind.LINE,
626
650
  height: Annotated[int, typer.Option("--height", min=3, help="Height in rows.")] = 15,
627
651
  bins: Annotated[int, typer.Option("--bins", min=1, help="Bins of a histogram.")] = 10,
652
+ resample: Annotated[
653
+ bool,
654
+ typer.Option(
655
+ "--resample/--no-resample",
656
+ help="Reduce the rows to the points the canvas can show (line and scatter only).",
657
+ ),
658
+ ] = True,
628
659
  x_label: Annotated[str | None, typer.Option("--x-label", help="Label of the X axis.")] = None,
629
660
  y_label: Annotated[str | None, typer.Option("--y-label", help="Label of the Y axis.")] = None,
630
661
  params: ParamOption = None,
@@ -635,12 +666,13 @@ def chart(
635
666
  parameters = _parameters(params)
636
667
  with _reporting_errors(), open_database(database) as db:
637
668
  text = sys.stdin.read() if sql == "-" else sql
638
- result = db.execute(text, parameters if parameters else ())
639
- if not result.returns_rows:
640
- _fail("the statement returned no rows")
669
+ bound = parameters if parameters else ()
641
670
  use_color = resolve_color(color)
642
671
  screen = width if width is not None else shutil.get_terminal_size().columns
643
672
  if kind is ChartKind.HIST:
673
+ result = db.execute(text, bound)
674
+ if not result.returns_rows:
675
+ _fail("the statement returned no rows")
644
676
  values, skipped = histogram_values(result)
645
677
  drawing = render_histogram(
646
678
  values,
@@ -652,7 +684,23 @@ def chart(
652
684
  y_label=y_label or "count",
653
685
  )
654
686
  else:
655
- series, skipped = series_from_result(result)
687
+ if resample:
688
+ # Reduces as it reads, so the rows never pile up in memory.
689
+ stream = db.stream(text, bound)
690
+ if not stream.returns_rows:
691
+ _fail("the statement returned no rows")
692
+ columns = stream.columns
693
+ series, skipped, read = stream_series(
694
+ stream, kind=kind, width=screen, height=height
695
+ )
696
+ if len(series[0].x) < read:
697
+ typer.echo(f"resampled {read} rows to {len(series[0].x)} points", err=True)
698
+ else:
699
+ result = db.execute(text, bound)
700
+ if not result.returns_rows:
701
+ _fail("the statement returned no rows")
702
+ columns = result.columns
703
+ series, skipped = series_from_result(result)
656
704
  default_y = series[0].label if len(series) == 1 else "value"
657
705
  drawing = render_chart(
658
706
  series,
@@ -660,7 +708,7 @@ def chart(
660
708
  width=screen,
661
709
  height=height,
662
710
  color=use_color,
663
- x_label=x_label or result.columns[0],
711
+ x_label=x_label or columns[0],
664
712
  y_label=y_label or default_y,
665
713
  )
666
714
  if skipped:
@@ -722,11 +770,21 @@ def export(
722
770
  with _reporting_errors(), open_database(database) as db:
723
771
  if every:
724
772
  assert output is not None
725
- output.mkdir(parents=True, exist_ok=True)
773
+ targets: dict[str, str] = {}
726
774
  for name in db.names():
727
- target = output / f"{name}.{_EXTENSIONS[output_format]}"
728
- with target.open("w", encoding="utf-8") as handle:
729
- write_rows(db.stream_rows(name), options, handle)
775
+ file_name = _export_file_name(name, _EXTENSIONS[output_format])
776
+ if file_name in targets:
777
+ _fail(f"{name} and {targets[file_name]} both export to {file_name}")
778
+ targets[file_name] = name
779
+ output.mkdir(parents=True, exist_ok=True)
780
+ for file_name, name in targets.items():
781
+ target = output / file_name
782
+ try:
783
+ with target.open("w", encoding="utf-8") as handle:
784
+ write_rows(db.stream_rows(name), options, handle)
785
+ except ExplorerError as error:
786
+ target.unlink(missing_ok=True)
787
+ typer.secho(f"skipped {name}: {error}", err=True, fg=typer.colors.YELLOW)
730
788
  return
731
789
  assert table is not None
732
790
  if output is None:
@@ -762,15 +820,17 @@ def import_(
762
820
  delimiter: Annotated[
763
821
  str | None, typer.Option("--delimiter", help="Field delimiter for CSV/TSV.")
764
822
  ] = None,
823
+ encoding: Annotated[
824
+ str, typer.Option("--encoding", help="Encoding of the file.")
825
+ ] = "utf-8-sig",
765
826
  ) -> None:
766
827
  """Load a CSV, TSV or JSON file into a table, creating it if needed."""
767
828
  input_format = output_format or _FORMAT_BY_SUFFIX.get(file.suffix.lower())
768
829
  if input_format is None:
769
830
  _fail(f"cannot tell the format of {file.name}; pass --format")
831
+ text = _read_text(file, encoding)
770
832
  with _reporting_errors(), open_database(database, write=True) as db:
771
- headers, raw_rows = parse_rows(
772
- file.read_text(encoding="utf-8-sig"), input_format, delimiter=delimiter
773
- )
833
+ headers, raw_rows = parse_rows(text, input_format, delimiter=delimiter)
774
834
  types = infer_types(raw_rows, len(headers))
775
835
  count = db.import_rows(
776
836
  table, list(zip(headers, types, strict=True)), coerce_rows(raw_rows, types)
@@ -498,7 +498,10 @@ class Explorer:
498
498
  table, columns=columns, where=where, order_by=order_by, descending=descending
499
499
  )
500
500
  query = f"SELECT {selection} {source}{order} LIMIT ? OFFSET ?"
501
- return self.stream(query, (-1 if limit is None else limit, offset))
501
+ try:
502
+ return self.stream(query, (-1 if limit is None else limit, offset))
503
+ except sqlite3.Error as error:
504
+ raise translate_error(error, write=self._write) from error
502
505
 
503
506
  def stats(
504
507
  self,
@@ -53,6 +53,11 @@ ELLIPSIS = "…"
53
53
  _ANSI = re.compile(r"\x1b\[[0-9;]*m")
54
54
  _INTEGER = re.compile(r"^[+-]?\d+$")
55
55
  _REAL = re.compile(r"^[+-]?(\d+\.\d*|\.\d+|\d+)([eE][+-]?\d+)?$")
56
+ _LEADING_ZERO = re.compile(r"^[+-]?0\d")
57
+ # A tuple, not bool | int: the union would be rebuilt on every call.
58
+ _INTEGRAL = (bool, int)
59
+ # Integers outside this range do not fit in a SQLite INTEGER column.
60
+ _INT64 = range(-(2**63), 2**63)
56
61
 
57
62
 
58
63
  class OutputFormat(str, Enum):
@@ -262,6 +267,11 @@ def default_page_size() -> int:
262
267
  return max(1, shutil.get_terminal_size().lines - 4)
263
268
 
264
269
 
270
+ def _page_footer(number: int, pages: int, rows: int) -> str:
271
+ plural = "" if rows == 1 else "s"
272
+ return f"page {number} of {pages} ({rows} row{plural})"
273
+
274
+
265
275
  def paginate(
266
276
  result: ResultSet, *, page: int | None = None, page_size: int | None = None
267
277
  ) -> tuple[ResultSet, str | None]:
@@ -275,7 +285,7 @@ def paginate(
275
285
  pages = max(1, -(-result.total // size))
276
286
  if number > pages:
277
287
  raise ExplorerError(f"page {number} is out of range (1-{pages})")
278
- return result, f"page {number} of {pages} ({result.total} rows)"
288
+ return result, _page_footer(number, pages, result.total)
279
289
  pages = max(1, -(-len(result.rows) // size))
280
290
  if number > pages:
281
291
  raise ExplorerError(f"page {number} is out of range (1-{pages})")
@@ -283,7 +293,7 @@ def paginate(
283
293
  sliced = ResultSet(
284
294
  columns=result.columns, rows=result.rows[start : start + size], rowcount=result.rowcount
285
295
  )
286
- return sliced, f"page {number} of {pages} ({len(result.rows)} rows)"
296
+ return sliced, _page_footer(number, pages, len(result.rows))
287
297
 
288
298
 
289
299
  def stdout_is_tty() -> bool:
@@ -422,15 +432,19 @@ def parse_rows(
422
432
  if fmt not in (OutputFormat.CSV, OutputFormat.TSV):
423
433
  raise ExplorerError(f"cannot import from the {fmt.value} format")
424
434
  separator = delimiter or ("\t" if fmt is OutputFormat.TSV else ",")
425
- reader = csv.reader(io.StringIO(text), delimiter=separator)
435
+ if len(separator) != 1:
436
+ raise ExplorerError("the delimiter must be a single character")
426
437
  try:
427
- headers = next(reader)
428
- except StopIteration:
429
- raise ExplorerError("empty file") from None
438
+ records = list(csv.reader(io.StringIO(text), delimiter=separator))
439
+ except csv.Error as error:
440
+ raise ExplorerError(f"cannot read the file: {error}") from error
441
+ if not records:
442
+ raise ExplorerError("empty file")
443
+ headers, *rest = records
430
444
  if not any(header.strip() for header in headers):
431
445
  raise ExplorerError("empty file")
432
446
  rows: list[list[object]] = []
433
- for number, record in enumerate(reader, start=2):
447
+ for number, record in enumerate(rest, start=2):
434
448
  if not record:
435
449
  continue
436
450
  if len(record) > len(headers):
@@ -459,19 +473,24 @@ def _parse_json_rows(text: str) -> tuple[list[str], list[list[object]]]:
459
473
  def _plain_json_value(value: object) -> object:
460
474
  if isinstance(value, bool):
461
475
  return int(value)
476
+ if isinstance(value, int) and value not in _INT64:
477
+ return str(value)
462
478
  if isinstance(value, (dict, list)):
463
479
  return json.dumps(value, ensure_ascii=False)
464
480
  return value
465
481
 
466
482
 
467
483
  def _classify(value: object) -> str:
468
- if isinstance(value, bool | int):
484
+ if isinstance(value, _INTEGRAL):
469
485
  return "INTEGER"
470
486
  if isinstance(value, float):
471
487
  return "REAL"
472
488
  text = str(value).strip()
489
+ if _LEADING_ZERO.match(text):
490
+ # 007 is a code, not a number: storing it as one would drop the zeros.
491
+ return "TEXT"
473
492
  if _INTEGER.match(text):
474
- return "INTEGER"
493
+ return "INTEGER" if int(text) in _INT64 else "TEXT"
475
494
  if _REAL.match(text):
476
495
  return "REAL"
477
496
  return "TEXT"