sqlitexplorer 1.0.0__tar.gz → 1.1.2__tar.gz
This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
- sqlitexplorer-1.1.2/MANIFEST.in +2 -0
- {sqlitexplorer-1.0.0/sqlitexplorer.egg-info → sqlitexplorer-1.1.2}/PKG-INFO +22 -24
- {sqlitexplorer-1.0.0 → sqlitexplorer-1.1.2}/README.md +20 -23
- {sqlitexplorer-1.0.0 → sqlitexplorer-1.1.2}/pyproject.toml +2 -1
- {sqlitexplorer-1.0.0 → sqlitexplorer-1.1.2}/sqlitexplorer/charts.py +90 -13
- {sqlitexplorer-1.0.0 → sqlitexplorer-1.1.2}/sqlitexplorer/cli.py +75 -15
- {sqlitexplorer-1.0.0 → sqlitexplorer-1.1.2}/sqlitexplorer/core.py +4 -1
- {sqlitexplorer-1.0.0 → sqlitexplorer-1.1.2}/sqlitexplorer/render.py +28 -9
- {sqlitexplorer-1.0.0 → sqlitexplorer-1.1.2/sqlitexplorer.egg-info}/PKG-INFO +22 -24
- {sqlitexplorer-1.0.0 → sqlitexplorer-1.1.2}/sqlitexplorer.egg-info/SOURCES.txt +2 -0
- {sqlitexplorer-1.0.0 → sqlitexplorer-1.1.2}/sqlitexplorer.egg-info/requires.txt +1 -0
- sqlitexplorer-1.1.2/tests/conftest.py +83 -0
- sqlitexplorer-1.1.2/tests/test_charts.py +229 -0
- {sqlitexplorer-1.0.0 → sqlitexplorer-1.1.2}/tests/test_cli.py +124 -1
- {sqlitexplorer-1.0.0 → sqlitexplorer-1.1.2}/tests/test_core.py +12 -0
- {sqlitexplorer-1.0.0 → sqlitexplorer-1.1.2}/tests/test_render.py +28 -0
- sqlitexplorer-1.0.0/tests/test_charts.py +0 -108
- {sqlitexplorer-1.0.0 → sqlitexplorer-1.1.2}/LICENSE +0 -0
- {sqlitexplorer-1.0.0 → sqlitexplorer-1.1.2}/setup.cfg +0 -0
- {sqlitexplorer-1.0.0 → sqlitexplorer-1.1.2}/sqlitexplorer/__init__.py +0 -0
- {sqlitexplorer-1.0.0 → sqlitexplorer-1.1.2}/sqlitexplorer/__main__.py +0 -0
- {sqlitexplorer-1.0.0 → sqlitexplorer-1.1.2}/sqlitexplorer/completion.py +0 -0
- {sqlitexplorer-1.0.0 → sqlitexplorer-1.1.2}/sqlitexplorer/shell.py +0 -0
- {sqlitexplorer-1.0.0 → sqlitexplorer-1.1.2}/sqlitexplorer.egg-info/dependency_links.txt +0 -0
- {sqlitexplorer-1.0.0 → sqlitexplorer-1.1.2}/sqlitexplorer.egg-info/entry_points.txt +0 -0
- {sqlitexplorer-1.0.0 → sqlitexplorer-1.1.2}/sqlitexplorer.egg-info/top_level.txt +0 -0
- {sqlitexplorer-1.0.0 → sqlitexplorer-1.1.2}/tests/test_completion.py +0 -0
- {sqlitexplorer-1.0.0 → sqlitexplorer-1.1.2}/tests/test_shell.py +0 -0
|
@@ -1,6 +1,6 @@
|
|
|
1
1
|
Metadata-Version: 2.4
|
|
2
2
|
Name: sqlitexplorer
|
|
3
|
-
Version: 1.
|
|
3
|
+
Version: 1.1.2
|
|
4
4
|
Summary: Explore SQLite databases from the terminal: tables, schema, stats, search, queries, charts, export/import and an interactive shell.
|
|
5
5
|
Author-email: "Carlos A. Planchón" <carlosandresplanchonprestes@gmail.com>
|
|
6
6
|
License-Expression: MIT
|
|
@@ -21,6 +21,7 @@ Description-Content-Type: text/markdown
|
|
|
21
21
|
License-File: LICENSE
|
|
22
22
|
Requires-Dist: outfancy>=0.11
|
|
23
23
|
Requires-Dist: plotille>=6
|
|
24
|
+
Requires-Dist: plotilleresample>=1.0
|
|
24
25
|
Requires-Dist: typer>=0.12
|
|
25
26
|
Provides-Extra: dev
|
|
26
27
|
Requires-Dist: pytest>=8; extra == "dev"
|
|
@@ -35,7 +36,8 @@ A command-line explorer for SQLite databases. It lists tables, prints schemas,
|
|
|
35
36
|
dumps rows, computes statistics, searches values, runs ad-hoc queries, draws
|
|
36
37
|
charts, exports and imports data, and offers an interactive shell. Tables are
|
|
37
38
|
rendered with [outfancy](https://github.com/carlosplanchon/outfancy) and charts
|
|
38
|
-
with [plotille](https://github.com/tammoippen/plotille)
|
|
39
|
+
with [plotille](https://github.com/tammoippen/plotille), resampled by
|
|
40
|
+
[plotilleresample](https://github.com/carlosplanchon/plotilleresample).
|
|
39
41
|
|
|
40
42
|
[](https://github.com/carlosplanchon/sqlitexplorer/actions/workflows/ci.yml)
|
|
41
43
|
[](https://pypi.org/project/sqlitexplorer/)
|
|
@@ -127,6 +129,8 @@ Big results do not need to fit in memory: `--page` fetches only the requested
|
|
|
127
129
|
page (in SQL for `show`, from the cursor for `query`), and the csv, tsv, json
|
|
128
130
|
and markdown formats, `export` and `dump` are written row by row. The table
|
|
129
131
|
format is the exception, since it needs every row to size its columns.
|
|
132
|
+
`chart` reduces as it reads, so the memory it needs does not grow with the
|
|
133
|
+
table; `chart --no-resample` is the one that holds every row.
|
|
130
134
|
|
|
131
135
|
## Queries
|
|
132
136
|
|
|
@@ -159,6 +163,14 @@ stderr. `--kind` selects `line` (default), `scatter` or `hist` (first column
|
|
|
159
163
|
only, `--bins`). `--height`, `--width`, `--x-label`, `--y-label` and `--color`
|
|
160
164
|
adjust the drawing.
|
|
161
165
|
|
|
166
|
+
Line and scatter charts are reduced to what the canvas can show, keeping the
|
|
167
|
+
minimum and the maximum of every column of braille dots, so spikes survive and
|
|
168
|
+
the true extremes keep the X of their own row. The rows are reduced as they are
|
|
169
|
+
read, in chunks, so a chart over millions of rows needs no more memory than one
|
|
170
|
+
over a thousand; the reduction is reported on stderr. `--no-resample` reads and
|
|
171
|
+
plots every row instead. Histograms are never reduced, since dropping rows would
|
|
172
|
+
change the distribution.
|
|
173
|
+
|
|
162
174
|
## Export, import, dump and diff
|
|
163
175
|
|
|
164
176
|
```sh
|
|
@@ -170,8 +182,14 @@ sqlitexplorer diff app.db backup.db # exit status 1 when th
|
|
|
170
182
|
```
|
|
171
183
|
|
|
172
184
|
`import` creates the table when it does not exist, inferring INTEGER, REAL or
|
|
173
|
-
TEXT for each column;
|
|
174
|
-
|
|
185
|
+
TEXT for each column; values with leading zeros and integers too large for
|
|
186
|
+
SQLite stay TEXT, so codes such as `007` keep their digits. Empty CSV cells
|
|
187
|
+
become NULL. The format comes from the file extension unless `--format` is
|
|
188
|
+
given, and `--encoding` reads a file that is not UTF-8.
|
|
189
|
+
|
|
190
|
+
`export --all` writes one file per table and view inside the directory: a name
|
|
191
|
+
that would not be a valid file name is sanitised, and an object that cannot be
|
|
192
|
+
read (a view over a dropped table) is skipped with a warning on stderr.
|
|
175
193
|
|
|
176
194
|
## Shell
|
|
177
195
|
|
|
@@ -204,26 +222,6 @@ uv run pytest
|
|
|
204
222
|
uv run ruff check .
|
|
205
223
|
```
|
|
206
224
|
|
|
207
|
-
## Releasing
|
|
208
|
-
|
|
209
|
-
Releases are driven by version tags. Pushing `vX.Y.Z` runs the tests, builds
|
|
210
|
-
the distributions, publishes them to PyPI with
|
|
211
|
-
[trusted publishing](https://docs.pypi.org/trusted-publishers/) and creates a
|
|
212
|
-
GitHub release with the artifacts attached.
|
|
213
|
-
|
|
214
|
-
```sh
|
|
215
|
-
uv version 0.3.0 # or: uv version --bump minor
|
|
216
|
-
git commit -am "Release 0.3.0"
|
|
217
|
-
git tag v0.3.0
|
|
218
|
-
git push origin master v0.3.0
|
|
219
|
-
```
|
|
220
|
-
|
|
221
|
-
The tag must match the version in `pyproject.toml`; the workflow refuses to
|
|
222
|
-
publish otherwise. Before the first release, register the repository as a
|
|
223
|
-
trusted publisher of the project on PyPI with the workflow name `release.yml`
|
|
224
|
-
and the environment `pypi`, and create that environment in the repository
|
|
225
|
-
settings on GitHub.
|
|
226
|
-
|
|
227
225
|
## License
|
|
228
226
|
|
|
229
227
|
MIT. See [LICENSE](LICENSE).
|
|
@@ -6,7 +6,8 @@ A command-line explorer for SQLite databases. It lists tables, prints schemas,
|
|
|
6
6
|
dumps rows, computes statistics, searches values, runs ad-hoc queries, draws
|
|
7
7
|
charts, exports and imports data, and offers an interactive shell. Tables are
|
|
8
8
|
rendered with [outfancy](https://github.com/carlosplanchon/outfancy) and charts
|
|
9
|
-
with [plotille](https://github.com/tammoippen/plotille)
|
|
9
|
+
with [plotille](https://github.com/tammoippen/plotille), resampled by
|
|
10
|
+
[plotilleresample](https://github.com/carlosplanchon/plotilleresample).
|
|
10
11
|
|
|
11
12
|
[](https://github.com/carlosplanchon/sqlitexplorer/actions/workflows/ci.yml)
|
|
12
13
|
[](https://pypi.org/project/sqlitexplorer/)
|
|
@@ -98,6 +99,8 @@ Big results do not need to fit in memory: `--page` fetches only the requested
|
|
|
98
99
|
page (in SQL for `show`, from the cursor for `query`), and the csv, tsv, json
|
|
99
100
|
and markdown formats, `export` and `dump` are written row by row. The table
|
|
100
101
|
format is the exception, since it needs every row to size its columns.
|
|
102
|
+
`chart` reduces as it reads, so the memory it needs does not grow with the
|
|
103
|
+
table; `chart --no-resample` is the one that holds every row.
|
|
101
104
|
|
|
102
105
|
## Queries
|
|
103
106
|
|
|
@@ -130,6 +133,14 @@ stderr. `--kind` selects `line` (default), `scatter` or `hist` (first column
|
|
|
130
133
|
only, `--bins`). `--height`, `--width`, `--x-label`, `--y-label` and `--color`
|
|
131
134
|
adjust the drawing.
|
|
132
135
|
|
|
136
|
+
Line and scatter charts are reduced to what the canvas can show, keeping the
|
|
137
|
+
minimum and the maximum of every column of braille dots, so spikes survive and
|
|
138
|
+
the true extremes keep the X of their own row. The rows are reduced as they are
|
|
139
|
+
read, in chunks, so a chart over millions of rows needs no more memory than one
|
|
140
|
+
over a thousand; the reduction is reported on stderr. `--no-resample` reads and
|
|
141
|
+
plots every row instead. Histograms are never reduced, since dropping rows would
|
|
142
|
+
change the distribution.
|
|
143
|
+
|
|
133
144
|
## Export, import, dump and diff
|
|
134
145
|
|
|
135
146
|
```sh
|
|
@@ -141,8 +152,14 @@ sqlitexplorer diff app.db backup.db # exit status 1 when th
|
|
|
141
152
|
```
|
|
142
153
|
|
|
143
154
|
`import` creates the table when it does not exist, inferring INTEGER, REAL or
|
|
144
|
-
TEXT for each column;
|
|
145
|
-
|
|
155
|
+
TEXT for each column; values with leading zeros and integers too large for
|
|
156
|
+
SQLite stay TEXT, so codes such as `007` keep their digits. Empty CSV cells
|
|
157
|
+
become NULL. The format comes from the file extension unless `--format` is
|
|
158
|
+
given, and `--encoding` reads a file that is not UTF-8.
|
|
159
|
+
|
|
160
|
+
`export --all` writes one file per table and view inside the directory: a name
|
|
161
|
+
that would not be a valid file name is sanitised, and an object that cannot be
|
|
162
|
+
read (a view over a dropped table) is skipped with a warning on stderr.
|
|
146
163
|
|
|
147
164
|
## Shell
|
|
148
165
|
|
|
@@ -175,26 +192,6 @@ uv run pytest
|
|
|
175
192
|
uv run ruff check .
|
|
176
193
|
```
|
|
177
194
|
|
|
178
|
-
## Releasing
|
|
179
|
-
|
|
180
|
-
Releases are driven by version tags. Pushing `vX.Y.Z` runs the tests, builds
|
|
181
|
-
the distributions, publishes them to PyPI with
|
|
182
|
-
[trusted publishing](https://docs.pypi.org/trusted-publishers/) and creates a
|
|
183
|
-
GitHub release with the artifacts attached.
|
|
184
|
-
|
|
185
|
-
```sh
|
|
186
|
-
uv version 0.3.0 # or: uv version --bump minor
|
|
187
|
-
git commit -am "Release 0.3.0"
|
|
188
|
-
git tag v0.3.0
|
|
189
|
-
git push origin master v0.3.0
|
|
190
|
-
```
|
|
191
|
-
|
|
192
|
-
The tag must match the version in `pyproject.toml`; the workflow refuses to
|
|
193
|
-
publish otherwise. Before the first release, register the repository as a
|
|
194
|
-
trusted publisher of the project on PyPI with the workflow name `release.yml`
|
|
195
|
-
and the environment `pypi`, and create that environment in the repository
|
|
196
|
-
settings on GitHub.
|
|
197
|
-
|
|
198
195
|
## License
|
|
199
196
|
|
|
200
197
|
MIT. See [LICENSE](LICENSE).
|
|
@@ -4,7 +4,7 @@ build-backend = "setuptools.build_meta"
|
|
|
4
4
|
|
|
5
5
|
[project]
|
|
6
6
|
name = "sqlitexplorer"
|
|
7
|
-
version = "1.
|
|
7
|
+
version = "1.1.2"
|
|
8
8
|
description = "Explore SQLite databases from the terminal: tables, schema, stats, search, queries, charts, export/import and an interactive shell."
|
|
9
9
|
readme = "README.md"
|
|
10
10
|
license = "MIT"
|
|
@@ -29,6 +29,7 @@ classifiers = [
|
|
|
29
29
|
dependencies = [
|
|
30
30
|
"outfancy>=0.11",
|
|
31
31
|
"plotille>=6",
|
|
32
|
+
"plotilleresample>=1.0",
|
|
32
33
|
"typer>=0.12",
|
|
33
34
|
]
|
|
34
35
|
|
|
@@ -12,10 +12,12 @@ from contextlib import contextmanager
|
|
|
12
12
|
from dataclasses import dataclass
|
|
13
13
|
from datetime import datetime
|
|
14
14
|
from enum import Enum
|
|
15
|
+
from itertools import islice
|
|
15
16
|
|
|
16
17
|
import plotille
|
|
18
|
+
import plotilleresample
|
|
17
19
|
|
|
18
|
-
from sqlitexplorer.core import ExplorerError, ResultSet
|
|
20
|
+
from sqlitexplorer.core import ExplorerError, ResultSet, RowStream
|
|
19
21
|
|
|
20
22
|
__all__ = [
|
|
21
23
|
"ChartKind",
|
|
@@ -23,7 +25,9 @@ __all__ = [
|
|
|
23
25
|
"histogram_values",
|
|
24
26
|
"render_chart",
|
|
25
27
|
"render_histogram",
|
|
28
|
+
"resample_series",
|
|
26
29
|
"series_from_result",
|
|
30
|
+
"stream_series",
|
|
27
31
|
]
|
|
28
32
|
|
|
29
33
|
PALETTE = ("red", "green", "yellow", "blue", "magenta", "cyan")
|
|
@@ -31,6 +35,13 @@ PALETTE = ("red", "green", "yellow", "blue", "magenta", "cyan")
|
|
|
31
35
|
AXIS_LABEL_WIDTH = 8
|
|
32
36
|
# Characters taken by the Y axis (ticks, label and separator) next to the canvas.
|
|
33
37
|
AXIS_WIDTH = 12
|
|
38
|
+
# A tuple, not bool | int | float: the union would be rebuilt on every call,
|
|
39
|
+
# and this runs once per value of the result.
|
|
40
|
+
_NUMERIC = (bool, int, float)
|
|
41
|
+
# Rows read at a time when streaming, and how many reduced points may pile
|
|
42
|
+
# up before they are reduced again.
|
|
43
|
+
_CHUNK = 65536
|
|
44
|
+
_PILE = 8
|
|
34
45
|
|
|
35
46
|
|
|
36
47
|
class ChartKind(str, Enum):
|
|
@@ -47,7 +58,7 @@ class Series:
|
|
|
47
58
|
|
|
48
59
|
|
|
49
60
|
def _number(value: object) -> float | None:
|
|
50
|
-
if isinstance(value,
|
|
61
|
+
if isinstance(value, _NUMERIC):
|
|
51
62
|
return float(value)
|
|
52
63
|
if isinstance(value, str):
|
|
53
64
|
try:
|
|
@@ -69,11 +80,13 @@ def _x_value(value: object) -> float | datetime | None:
|
|
|
69
80
|
return None
|
|
70
81
|
|
|
71
82
|
|
|
72
|
-
def series_from_result(result: ResultSet) -> tuple[list[Series], int]:
|
|
83
|
+
def series_from_result(result: ResultSet, *, require_rows: bool = True) -> tuple[list[Series], int]:
|
|
73
84
|
"""Split *result* into one series per numeric column after the first.
|
|
74
85
|
|
|
75
86
|
Rows with a NULL in the X column or in any series are skipped; the second
|
|
76
|
-
item of the returned tuple counts them.
|
|
87
|
+
item of the returned tuple counts them. With *require_rows* false an empty
|
|
88
|
+
result yields empty series instead of raising, which is what the streaming
|
|
89
|
+
reader needs for a chunk that holds nothing usable.
|
|
77
90
|
"""
|
|
78
91
|
if len(result.columns) < 2:
|
|
79
92
|
raise ExplorerError("need an X column and at least one numeric column")
|
|
@@ -82,8 +95,10 @@ def series_from_result(result: ResultSet) -> tuple[list[Series], int]:
|
|
|
82
95
|
ys: list[list[float]] = [[] for _ in y_names]
|
|
83
96
|
x_type: type | None = None
|
|
84
97
|
skipped = 0
|
|
98
|
+
keep_x = xs.append
|
|
99
|
+
keep = [bucket.append for bucket in ys]
|
|
85
100
|
for row in result.rows:
|
|
86
|
-
if
|
|
101
|
+
if None in row:
|
|
87
102
|
skipped += 1
|
|
88
103
|
continue
|
|
89
104
|
x = _x_value(row[0])
|
|
@@ -93,26 +108,88 @@ def series_from_result(result: ResultSet) -> tuple[list[Series], int]:
|
|
|
93
108
|
x_type = type(x)
|
|
94
109
|
elif not isinstance(x, x_type):
|
|
95
110
|
raise ExplorerError(f"column {x_name} mixes numbers and dates")
|
|
96
|
-
|
|
97
|
-
for name, value in zip(y_names, row[1:], strict=True):
|
|
111
|
+
for name, value, append in zip(y_names, row[1:], keep, strict=True):
|
|
98
112
|
number = _number(value)
|
|
99
113
|
if number is None:
|
|
100
114
|
raise ExplorerError(f"column {name} is not numeric: {value!r}")
|
|
101
|
-
|
|
102
|
-
|
|
103
|
-
|
|
104
|
-
bucket.append(number)
|
|
105
|
-
if not xs:
|
|
115
|
+
append(number)
|
|
116
|
+
keep_x(x)
|
|
117
|
+
if require_rows and not xs:
|
|
106
118
|
raise ExplorerError("no rows to plot")
|
|
107
119
|
return [
|
|
108
120
|
Series(label=name, x=xs, y=bucket) for name, bucket in zip(y_names, ys, strict=True)
|
|
109
121
|
], skipped
|
|
110
122
|
|
|
111
123
|
|
|
124
|
+
def resample_series(
|
|
125
|
+
series: Sequence[Series], *, kind: ChartKind, width: int, height: int
|
|
126
|
+
) -> list[Series]:
|
|
127
|
+
"""Reduce every series to the points the canvas can actually draw.
|
|
128
|
+
|
|
129
|
+
min/max keeps the extremes of every bucket, so spikes survive, and it only
|
|
130
|
+
indexes X, which the LTTB resamplers cannot do when X is a date.
|
|
131
|
+
"""
|
|
132
|
+
budget = _canvas_width(width)
|
|
133
|
+
reduce = (
|
|
134
|
+
plotilleresample.resample_scatter
|
|
135
|
+
if kind is ChartKind.SCATTER
|
|
136
|
+
else plotilleresample.resample_plot_minmax
|
|
137
|
+
)
|
|
138
|
+
reduced = []
|
|
139
|
+
for item in series:
|
|
140
|
+
x, y = reduce(item.x, item.y, budget, height)
|
|
141
|
+
reduced.append(Series(label=item.label, x=list(x), y=list(y)))
|
|
142
|
+
return reduced
|
|
143
|
+
|
|
144
|
+
|
|
145
|
+
def stream_series(
|
|
146
|
+
stream: RowStream, *, kind: ChartKind, width: int, height: int
|
|
147
|
+
) -> tuple[list[Series], int, int]:
|
|
148
|
+
"""Reduce *stream* to the canvas without ever holding every row.
|
|
149
|
+
|
|
150
|
+
Rows are read in chunks, each chunk is reduced on its own and the reduced
|
|
151
|
+
points are reduced again as they pile up, so the memory a chart needs stops
|
|
152
|
+
growing with the size of the table. Returns the series, how many rows were
|
|
153
|
+
skipped for their NULLs and how many were read.
|
|
154
|
+
"""
|
|
155
|
+
reduced: list[Series] = []
|
|
156
|
+
x_type: type | None = None
|
|
157
|
+
skipped = rows_read = 0
|
|
158
|
+
while True:
|
|
159
|
+
rows = list(islice(stream.rows, _CHUNK))
|
|
160
|
+
if not rows:
|
|
161
|
+
break
|
|
162
|
+
rows_read += len(rows)
|
|
163
|
+
chunk, chunk_skipped = series_from_result(
|
|
164
|
+
ResultSet(columns=stream.columns, rows=rows), require_rows=False
|
|
165
|
+
)
|
|
166
|
+
skipped += chunk_skipped
|
|
167
|
+
if not chunk[0].x:
|
|
168
|
+
continue
|
|
169
|
+
if x_type is None:
|
|
170
|
+
x_type = type(chunk[0].x[0])
|
|
171
|
+
elif not isinstance(chunk[0].x[0], x_type):
|
|
172
|
+
raise ExplorerError(f"column {stream.columns[0]} mixes numbers and dates")
|
|
173
|
+
chunk = resample_series(chunk, kind=kind, width=width, height=height)
|
|
174
|
+
reduced = (
|
|
175
|
+
[
|
|
176
|
+
Series(label=old.label, x=old.x + new.x, y=old.y + new.y)
|
|
177
|
+
for old, new in zip(reduced, chunk, strict=True)
|
|
178
|
+
]
|
|
179
|
+
if reduced
|
|
180
|
+
else chunk
|
|
181
|
+
)
|
|
182
|
+
if len(reduced[0].x) > _PILE * len(chunk[0].x):
|
|
183
|
+
reduced = resample_series(reduced, kind=kind, width=width, height=height)
|
|
184
|
+
if not reduced or not reduced[0].x:
|
|
185
|
+
raise ExplorerError("no rows to plot")
|
|
186
|
+
return resample_series(reduced, kind=kind, width=width, height=height), skipped, rows_read
|
|
187
|
+
|
|
188
|
+
|
|
112
189
|
def histogram_values(result: ResultSet) -> tuple[list[float], int]:
|
|
113
190
|
"""Numeric values of the first column of *result*, and how many NULLs were skipped."""
|
|
114
191
|
if not result.columns:
|
|
115
|
-
raise ExplorerError("
|
|
192
|
+
raise ExplorerError("need a numeric column to plot")
|
|
116
193
|
name = result.columns[0]
|
|
117
194
|
values: list[float] = []
|
|
118
195
|
skipped = 0
|
|
@@ -11,6 +11,7 @@ from __future__ import annotations
|
|
|
11
11
|
import difflib
|
|
12
12
|
import functools
|
|
13
13
|
import inspect
|
|
14
|
+
import re
|
|
14
15
|
import shutil
|
|
15
16
|
import sys
|
|
16
17
|
import time
|
|
@@ -28,6 +29,7 @@ from sqlitexplorer.charts import (
|
|
|
28
29
|
render_chart,
|
|
29
30
|
render_histogram,
|
|
30
31
|
series_from_result,
|
|
32
|
+
stream_series,
|
|
31
33
|
)
|
|
32
34
|
from sqlitexplorer.completion import complete_table
|
|
33
35
|
from sqlitexplorer.core import (
|
|
@@ -156,6 +158,7 @@ _EXTENSIONS = {
|
|
|
156
158
|
OutputFormat.MARKDOWN: "md",
|
|
157
159
|
}
|
|
158
160
|
_FORMAT_BY_SUFFIX = {".csv": OutputFormat.CSV, ".tsv": OutputFormat.TSV, ".json": OutputFormat.JSON}
|
|
161
|
+
_UNSAFE_IN_NAME = re.compile(r"[^\w.-]")
|
|
159
162
|
|
|
160
163
|
|
|
161
164
|
# --- Helpers ------------------------------------------------------------------
|
|
@@ -273,11 +276,31 @@ def _attach_all(db: Explorer, values: Sequence[str] | None, *, write: bool) -> N
|
|
|
273
276
|
db.attach(alias, path, write=write)
|
|
274
277
|
|
|
275
278
|
|
|
279
|
+
def _read_text(file: Path, encoding: str) -> str:
|
|
280
|
+
"""Read *file* as text, reporting a wrong encoding as a plain message."""
|
|
281
|
+
try:
|
|
282
|
+
return file.read_text(encoding=encoding)
|
|
283
|
+
except UnicodeDecodeError:
|
|
284
|
+
_fail(f"{file.name} is not valid {encoding} text")
|
|
285
|
+
except LookupError:
|
|
286
|
+
_fail(f"unknown encoding: {encoding}")
|
|
287
|
+
except OSError as error:
|
|
288
|
+
_fail(str(error))
|
|
289
|
+
|
|
290
|
+
|
|
291
|
+
def _export_file_name(name: str, extension: str) -> str:
|
|
292
|
+
"""A file name for *name* that cannot escape the directory it is written in."""
|
|
293
|
+
safe = _UNSAFE_IN_NAME.sub("_", name).lstrip(".")
|
|
294
|
+
return f"{safe or '_'}.{extension}"
|
|
295
|
+
|
|
296
|
+
|
|
276
297
|
def _read_sql(sql: str | None, file: Path | None) -> list[str]:
|
|
277
|
-
if
|
|
298
|
+
if sql is not None and file is not None:
|
|
278
299
|
_fail("give either an SQL statement or --file, not both")
|
|
300
|
+
if sql is None and file is None:
|
|
301
|
+
_fail("give an SQL statement, - to read it from stdin, or --file")
|
|
279
302
|
if file is not None:
|
|
280
|
-
text = file
|
|
303
|
+
text = _read_text(file, "utf-8")
|
|
281
304
|
elif sql == "-":
|
|
282
305
|
text = sys.stdin.read()
|
|
283
306
|
else:
|
|
@@ -617,7 +640,8 @@ def chart(
|
|
|
617
640
|
str,
|
|
618
641
|
typer.Argument(
|
|
619
642
|
show_default=False,
|
|
620
|
-
help="Query whose first column is X and the other numeric columns are series
|
|
643
|
+
help="Query whose first column is X and the other numeric columns are series,"
|
|
644
|
+
" or - to read it from stdin.",
|
|
621
645
|
),
|
|
622
646
|
],
|
|
623
647
|
kind: Annotated[
|
|
@@ -625,6 +649,13 @@ def chart(
|
|
|
625
649
|
] = ChartKind.LINE,
|
|
626
650
|
height: Annotated[int, typer.Option("--height", min=3, help="Height in rows.")] = 15,
|
|
627
651
|
bins: Annotated[int, typer.Option("--bins", min=1, help="Bins of a histogram.")] = 10,
|
|
652
|
+
resample: Annotated[
|
|
653
|
+
bool,
|
|
654
|
+
typer.Option(
|
|
655
|
+
"--resample/--no-resample",
|
|
656
|
+
help="Reduce the rows to the points the canvas can show (line and scatter only).",
|
|
657
|
+
),
|
|
658
|
+
] = True,
|
|
628
659
|
x_label: Annotated[str | None, typer.Option("--x-label", help="Label of the X axis.")] = None,
|
|
629
660
|
y_label: Annotated[str | None, typer.Option("--y-label", help="Label of the Y axis.")] = None,
|
|
630
661
|
params: ParamOption = None,
|
|
@@ -635,12 +666,13 @@ def chart(
|
|
|
635
666
|
parameters = _parameters(params)
|
|
636
667
|
with _reporting_errors(), open_database(database) as db:
|
|
637
668
|
text = sys.stdin.read() if sql == "-" else sql
|
|
638
|
-
|
|
639
|
-
if not result.returns_rows:
|
|
640
|
-
_fail("the statement returned no rows")
|
|
669
|
+
bound = parameters if parameters else ()
|
|
641
670
|
use_color = resolve_color(color)
|
|
642
671
|
screen = width if width is not None else shutil.get_terminal_size().columns
|
|
643
672
|
if kind is ChartKind.HIST:
|
|
673
|
+
result = db.execute(text, bound)
|
|
674
|
+
if not result.returns_rows:
|
|
675
|
+
_fail("the statement returned no rows")
|
|
644
676
|
values, skipped = histogram_values(result)
|
|
645
677
|
drawing = render_histogram(
|
|
646
678
|
values,
|
|
@@ -652,7 +684,23 @@ def chart(
|
|
|
652
684
|
y_label=y_label or "count",
|
|
653
685
|
)
|
|
654
686
|
else:
|
|
655
|
-
|
|
687
|
+
if resample:
|
|
688
|
+
# Reduces as it reads, so the rows never pile up in memory.
|
|
689
|
+
stream = db.stream(text, bound)
|
|
690
|
+
if not stream.returns_rows:
|
|
691
|
+
_fail("the statement returned no rows")
|
|
692
|
+
columns = stream.columns
|
|
693
|
+
series, skipped, read = stream_series(
|
|
694
|
+
stream, kind=kind, width=screen, height=height
|
|
695
|
+
)
|
|
696
|
+
if len(series[0].x) < read:
|
|
697
|
+
typer.echo(f"resampled {read} rows to {len(series[0].x)} points", err=True)
|
|
698
|
+
else:
|
|
699
|
+
result = db.execute(text, bound)
|
|
700
|
+
if not result.returns_rows:
|
|
701
|
+
_fail("the statement returned no rows")
|
|
702
|
+
columns = result.columns
|
|
703
|
+
series, skipped = series_from_result(result)
|
|
656
704
|
default_y = series[0].label if len(series) == 1 else "value"
|
|
657
705
|
drawing = render_chart(
|
|
658
706
|
series,
|
|
@@ -660,7 +708,7 @@ def chart(
|
|
|
660
708
|
width=screen,
|
|
661
709
|
height=height,
|
|
662
710
|
color=use_color,
|
|
663
|
-
x_label=x_label or
|
|
711
|
+
x_label=x_label or columns[0],
|
|
664
712
|
y_label=y_label or default_y,
|
|
665
713
|
)
|
|
666
714
|
if skipped:
|
|
@@ -722,11 +770,21 @@ def export(
|
|
|
722
770
|
with _reporting_errors(), open_database(database) as db:
|
|
723
771
|
if every:
|
|
724
772
|
assert output is not None
|
|
725
|
-
|
|
773
|
+
targets: dict[str, str] = {}
|
|
726
774
|
for name in db.names():
|
|
727
|
-
|
|
728
|
-
|
|
729
|
-
|
|
775
|
+
file_name = _export_file_name(name, _EXTENSIONS[output_format])
|
|
776
|
+
if file_name in targets:
|
|
777
|
+
_fail(f"{name} and {targets[file_name]} both export to {file_name}")
|
|
778
|
+
targets[file_name] = name
|
|
779
|
+
output.mkdir(parents=True, exist_ok=True)
|
|
780
|
+
for file_name, name in targets.items():
|
|
781
|
+
target = output / file_name
|
|
782
|
+
try:
|
|
783
|
+
with target.open("w", encoding="utf-8") as handle:
|
|
784
|
+
write_rows(db.stream_rows(name), options, handle)
|
|
785
|
+
except ExplorerError as error:
|
|
786
|
+
target.unlink(missing_ok=True)
|
|
787
|
+
typer.secho(f"skipped {name}: {error}", err=True, fg=typer.colors.YELLOW)
|
|
730
788
|
return
|
|
731
789
|
assert table is not None
|
|
732
790
|
if output is None:
|
|
@@ -762,15 +820,17 @@ def import_(
|
|
|
762
820
|
delimiter: Annotated[
|
|
763
821
|
str | None, typer.Option("--delimiter", help="Field delimiter for CSV/TSV.")
|
|
764
822
|
] = None,
|
|
823
|
+
encoding: Annotated[
|
|
824
|
+
str, typer.Option("--encoding", help="Encoding of the file.")
|
|
825
|
+
] = "utf-8-sig",
|
|
765
826
|
) -> None:
|
|
766
827
|
"""Load a CSV, TSV or JSON file into a table, creating it if needed."""
|
|
767
828
|
input_format = output_format or _FORMAT_BY_SUFFIX.get(file.suffix.lower())
|
|
768
829
|
if input_format is None:
|
|
769
830
|
_fail(f"cannot tell the format of {file.name}; pass --format")
|
|
831
|
+
text = _read_text(file, encoding)
|
|
770
832
|
with _reporting_errors(), open_database(database, write=True) as db:
|
|
771
|
-
headers, raw_rows = parse_rows(
|
|
772
|
-
file.read_text(encoding="utf-8-sig"), input_format, delimiter=delimiter
|
|
773
|
-
)
|
|
833
|
+
headers, raw_rows = parse_rows(text, input_format, delimiter=delimiter)
|
|
774
834
|
types = infer_types(raw_rows, len(headers))
|
|
775
835
|
count = db.import_rows(
|
|
776
836
|
table, list(zip(headers, types, strict=True)), coerce_rows(raw_rows, types)
|
|
@@ -498,7 +498,10 @@ class Explorer:
|
|
|
498
498
|
table, columns=columns, where=where, order_by=order_by, descending=descending
|
|
499
499
|
)
|
|
500
500
|
query = f"SELECT {selection} {source}{order} LIMIT ? OFFSET ?"
|
|
501
|
-
|
|
501
|
+
try:
|
|
502
|
+
return self.stream(query, (-1 if limit is None else limit, offset))
|
|
503
|
+
except sqlite3.Error as error:
|
|
504
|
+
raise translate_error(error, write=self._write) from error
|
|
502
505
|
|
|
503
506
|
def stats(
|
|
504
507
|
self,
|
|
@@ -53,6 +53,11 @@ ELLIPSIS = "…"
|
|
|
53
53
|
_ANSI = re.compile(r"\x1b\[[0-9;]*m")
|
|
54
54
|
_INTEGER = re.compile(r"^[+-]?\d+$")
|
|
55
55
|
_REAL = re.compile(r"^[+-]?(\d+\.\d*|\.\d+|\d+)([eE][+-]?\d+)?$")
|
|
56
|
+
_LEADING_ZERO = re.compile(r"^[+-]?0\d")
|
|
57
|
+
# A tuple, not bool | int: the union would be rebuilt on every call.
|
|
58
|
+
_INTEGRAL = (bool, int)
|
|
59
|
+
# Integers outside this range do not fit in a SQLite INTEGER column.
|
|
60
|
+
_INT64 = range(-(2**63), 2**63)
|
|
56
61
|
|
|
57
62
|
|
|
58
63
|
class OutputFormat(str, Enum):
|
|
@@ -262,6 +267,11 @@ def default_page_size() -> int:
|
|
|
262
267
|
return max(1, shutil.get_terminal_size().lines - 4)
|
|
263
268
|
|
|
264
269
|
|
|
270
|
+
def _page_footer(number: int, pages: int, rows: int) -> str:
|
|
271
|
+
plural = "" if rows == 1 else "s"
|
|
272
|
+
return f"page {number} of {pages} ({rows} row{plural})"
|
|
273
|
+
|
|
274
|
+
|
|
265
275
|
def paginate(
|
|
266
276
|
result: ResultSet, *, page: int | None = None, page_size: int | None = None
|
|
267
277
|
) -> tuple[ResultSet, str | None]:
|
|
@@ -275,7 +285,7 @@ def paginate(
|
|
|
275
285
|
pages = max(1, -(-result.total // size))
|
|
276
286
|
if number > pages:
|
|
277
287
|
raise ExplorerError(f"page {number} is out of range (1-{pages})")
|
|
278
|
-
return result,
|
|
288
|
+
return result, _page_footer(number, pages, result.total)
|
|
279
289
|
pages = max(1, -(-len(result.rows) // size))
|
|
280
290
|
if number > pages:
|
|
281
291
|
raise ExplorerError(f"page {number} is out of range (1-{pages})")
|
|
@@ -283,7 +293,7 @@ def paginate(
|
|
|
283
293
|
sliced = ResultSet(
|
|
284
294
|
columns=result.columns, rows=result.rows[start : start + size], rowcount=result.rowcount
|
|
285
295
|
)
|
|
286
|
-
return sliced,
|
|
296
|
+
return sliced, _page_footer(number, pages, len(result.rows))
|
|
287
297
|
|
|
288
298
|
|
|
289
299
|
def stdout_is_tty() -> bool:
|
|
@@ -422,15 +432,19 @@ def parse_rows(
|
|
|
422
432
|
if fmt not in (OutputFormat.CSV, OutputFormat.TSV):
|
|
423
433
|
raise ExplorerError(f"cannot import from the {fmt.value} format")
|
|
424
434
|
separator = delimiter or ("\t" if fmt is OutputFormat.TSV else ",")
|
|
425
|
-
|
|
435
|
+
if len(separator) != 1:
|
|
436
|
+
raise ExplorerError("the delimiter must be a single character")
|
|
426
437
|
try:
|
|
427
|
-
|
|
428
|
-
except
|
|
429
|
-
raise ExplorerError("
|
|
438
|
+
records = list(csv.reader(io.StringIO(text), delimiter=separator))
|
|
439
|
+
except csv.Error as error:
|
|
440
|
+
raise ExplorerError(f"cannot read the file: {error}") from error
|
|
441
|
+
if not records:
|
|
442
|
+
raise ExplorerError("empty file")
|
|
443
|
+
headers, *rest = records
|
|
430
444
|
if not any(header.strip() for header in headers):
|
|
431
445
|
raise ExplorerError("empty file")
|
|
432
446
|
rows: list[list[object]] = []
|
|
433
|
-
for number, record in enumerate(
|
|
447
|
+
for number, record in enumerate(rest, start=2):
|
|
434
448
|
if not record:
|
|
435
449
|
continue
|
|
436
450
|
if len(record) > len(headers):
|
|
@@ -459,19 +473,24 @@ def _parse_json_rows(text: str) -> tuple[list[str], list[list[object]]]:
|
|
|
459
473
|
def _plain_json_value(value: object) -> object:
|
|
460
474
|
if isinstance(value, bool):
|
|
461
475
|
return int(value)
|
|
476
|
+
if isinstance(value, int) and value not in _INT64:
|
|
477
|
+
return str(value)
|
|
462
478
|
if isinstance(value, (dict, list)):
|
|
463
479
|
return json.dumps(value, ensure_ascii=False)
|
|
464
480
|
return value
|
|
465
481
|
|
|
466
482
|
|
|
467
483
|
def _classify(value: object) -> str:
|
|
468
|
-
if isinstance(value,
|
|
484
|
+
if isinstance(value, _INTEGRAL):
|
|
469
485
|
return "INTEGER"
|
|
470
486
|
if isinstance(value, float):
|
|
471
487
|
return "REAL"
|
|
472
488
|
text = str(value).strip()
|
|
489
|
+
if _LEADING_ZERO.match(text):
|
|
490
|
+
# 007 is a code, not a number: storing it as one would drop the zeros.
|
|
491
|
+
return "TEXT"
|
|
473
492
|
if _INTEGER.match(text):
|
|
474
|
-
return "INTEGER"
|
|
493
|
+
return "INTEGER" if int(text) in _INT64 else "TEXT"
|
|
475
494
|
if _REAL.match(text):
|
|
476
495
|
return "REAL"
|
|
477
496
|
return "TEXT"
|