sqlitexplorer 1.1.3__tar.gz → 1.2.0__tar.gz

This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
Files changed (27) hide show
  1. {sqlitexplorer-1.1.3/sqlitexplorer.egg-info → sqlitexplorer-1.2.0}/PKG-INFO +19 -12
  2. {sqlitexplorer-1.1.3 → sqlitexplorer-1.2.0}/README.md +17 -10
  3. {sqlitexplorer-1.1.3 → sqlitexplorer-1.2.0}/pyproject.toml +2 -2
  4. {sqlitexplorer-1.1.3 → sqlitexplorer-1.2.0}/sqlitexplorer/charts.py +144 -61
  5. {sqlitexplorer-1.1.3 → sqlitexplorer-1.2.0}/sqlitexplorer/cli.py +27 -13
  6. {sqlitexplorer-1.1.3 → sqlitexplorer-1.2.0}/sqlitexplorer/core.py +16 -5
  7. {sqlitexplorer-1.1.3 → sqlitexplorer-1.2.0}/sqlitexplorer/render.py +22 -11
  8. {sqlitexplorer-1.1.3 → sqlitexplorer-1.2.0/sqlitexplorer.egg-info}/PKG-INFO +19 -12
  9. {sqlitexplorer-1.1.3 → sqlitexplorer-1.2.0}/sqlitexplorer.egg-info/requires.txt +1 -1
  10. {sqlitexplorer-1.1.3 → sqlitexplorer-1.2.0}/tests/test_charts.py +101 -26
  11. {sqlitexplorer-1.1.3 → sqlitexplorer-1.2.0}/tests/test_cli.py +42 -0
  12. {sqlitexplorer-1.1.3 → sqlitexplorer-1.2.0}/tests/test_core.py +13 -0
  13. {sqlitexplorer-1.1.3 → sqlitexplorer-1.2.0}/tests/test_render.py +9 -0
  14. {sqlitexplorer-1.1.3 → sqlitexplorer-1.2.0}/LICENSE +0 -0
  15. {sqlitexplorer-1.1.3 → sqlitexplorer-1.2.0}/MANIFEST.in +0 -0
  16. {sqlitexplorer-1.1.3 → sqlitexplorer-1.2.0}/setup.cfg +0 -0
  17. {sqlitexplorer-1.1.3 → sqlitexplorer-1.2.0}/sqlitexplorer/__init__.py +0 -0
  18. {sqlitexplorer-1.1.3 → sqlitexplorer-1.2.0}/sqlitexplorer/__main__.py +0 -0
  19. {sqlitexplorer-1.1.3 → sqlitexplorer-1.2.0}/sqlitexplorer/completion.py +0 -0
  20. {sqlitexplorer-1.1.3 → sqlitexplorer-1.2.0}/sqlitexplorer/shell.py +0 -0
  21. {sqlitexplorer-1.1.3 → sqlitexplorer-1.2.0}/sqlitexplorer.egg-info/SOURCES.txt +0 -0
  22. {sqlitexplorer-1.1.3 → sqlitexplorer-1.2.0}/sqlitexplorer.egg-info/dependency_links.txt +0 -0
  23. {sqlitexplorer-1.1.3 → sqlitexplorer-1.2.0}/sqlitexplorer.egg-info/entry_points.txt +0 -0
  24. {sqlitexplorer-1.1.3 → sqlitexplorer-1.2.0}/sqlitexplorer.egg-info/top_level.txt +0 -0
  25. {sqlitexplorer-1.1.3 → sqlitexplorer-1.2.0}/tests/conftest.py +0 -0
  26. {sqlitexplorer-1.1.3 → sqlitexplorer-1.2.0}/tests/test_completion.py +0 -0
  27. {sqlitexplorer-1.1.3 → sqlitexplorer-1.2.0}/tests/test_shell.py +0 -0
@@ -1,6 +1,6 @@
1
1
  Metadata-Version: 2.4
2
2
  Name: sqlitexplorer
3
- Version: 1.1.3
3
+ Version: 1.2.0
4
4
  Summary: Explore SQLite databases from the terminal: tables, schema, stats, search, queries, charts, export/import and an interactive shell.
5
5
  Author-email: "Carlos A. Planchón" <carlosandresplanchonprestes@gmail.com>
6
6
  License-Expression: MIT
@@ -21,7 +21,7 @@ Description-Content-Type: text/markdown
21
21
  License-File: LICENSE
22
22
  Requires-Dist: outfancy>=0.11
23
23
  Requires-Dist: plotille>=6
24
- Requires-Dist: plotilleresample>=1.0
24
+ Requires-Dist: plotilleresample>=1.2
25
25
  Requires-Dist: typer>=0.12
26
26
  Provides-Extra: dev
27
27
  Requires-Dist: pytest>=8; extra == "dev"
@@ -50,10 +50,11 @@ with [plotille](https://github.com/tammoippen/plotille), resampled by
50
50
  Requires Python 3.10 or newer.
51
51
 
52
52
  ```sh
53
- # From a clone of this repository:
54
- uv tool install .
53
+ uv tool install sqlitexplorer
55
54
  # or with pip:
56
- pip install .
55
+ pip install sqlitexplorer
56
+ # or from a clone of this repository:
57
+ uv tool install .
57
58
  ```
58
59
 
59
60
  ## Quick tour
@@ -129,8 +130,9 @@ Big results do not need to fit in memory: `--page` fetches only the requested
129
130
  page (in SQL for `show`, from the cursor for `query`), and the csv, tsv, json
130
131
  and markdown formats, `export` and `dump` are written row by row. The table
131
132
  format is the exception, since it needs every row to size its columns.
132
- `chart` reduces as it reads, so the memory it needs does not grow with the
133
- table; `chart --no-resample` is the one that holds every row.
133
+ `chart` reduces as it reads, and a histogram counts as it reads, so the memory
134
+ they need does not grow with the table; `chart --no-resample` is the one that
135
+ holds every row.
134
136
 
135
137
  ## Queries
136
138
 
@@ -161,15 +163,18 @@ The first column is the X axis (numbers or ISO dates), every other column is a
161
163
  series named after the column; rows with NULLs are skipped and counted on
162
164
  stderr. `--kind` selects `line` (default), `scatter` or `hist` (first column
163
165
  only, `--bins`). `--height`, `--width`, `--x-label`, `--y-label` and `--color`
164
- adjust the drawing.
166
+ adjust the drawing; the Y label is cut to eight characters, the room plotille
167
+ gives it.
165
168
 
166
169
  Line and scatter charts are reduced to what the canvas can show, keeping the
167
170
  minimum and the maximum of every column of braille dots, so spikes survive and
168
- the true extremes keep the X of their own row. The rows are reduced as they are
171
+ the true extremes keep the X of their own row; a scatter plot keeps a uniform
172
+ sample of its points as well, for its density. The rows are reduced as they are
169
173
  read, in chunks, so a chart over millions of rows needs no more memory than one
170
174
  over a thousand; the reduction is reported on stderr. `--no-resample` reads and
171
175
  plots every row instead. Histograms are never reduced, since dropping rows would
172
- change the distribution.
176
+ change the distribution: the query runs twice, once for the range and once for
177
+ the counts, so they do not hold the rows either.
173
178
 
174
179
  ## Export, import, dump and diff
175
180
 
@@ -183,8 +188,10 @@ sqlitexplorer diff app.db backup.db # exit status 1 when th
183
188
 
184
189
  `import` creates the table when it does not exist, inferring INTEGER, REAL or
185
190
  TEXT for each column; values with leading zeros and integers too large for
186
- SQLite stay TEXT, so codes such as `007` keep their digits. Empty CSV cells
187
- become NULL. The format comes from the file extension unless `--format` is
191
+ SQLite stay TEXT, so codes such as `007` keep their digits. From JSON the types
192
+ are the JSON types: numbers become INTEGER or REAL and strings stay TEXT even
193
+ when they look like numbers, so a json export imports back as it was. Empty CSV
194
+ cells become NULL. The format comes from the file extension unless `--format` is
188
195
  given, and `--encoding` reads a file that is not UTF-8.
189
196
 
190
197
  `export --all` writes one file per table and view inside the directory: a name
@@ -20,10 +20,11 @@ with [plotille](https://github.com/tammoippen/plotille), resampled by
20
20
  Requires Python 3.10 or newer.
21
21
 
22
22
  ```sh
23
- # From a clone of this repository:
24
- uv tool install .
23
+ uv tool install sqlitexplorer
25
24
  # or with pip:
26
- pip install .
25
+ pip install sqlitexplorer
26
+ # or from a clone of this repository:
27
+ uv tool install .
27
28
  ```
28
29
 
29
30
  ## Quick tour
@@ -99,8 +100,9 @@ Big results do not need to fit in memory: `--page` fetches only the requested
99
100
  page (in SQL for `show`, from the cursor for `query`), and the csv, tsv, json
100
101
  and markdown formats, `export` and `dump` are written row by row. The table
101
102
  format is the exception, since it needs every row to size its columns.
102
- `chart` reduces as it reads, so the memory it needs does not grow with the
103
- table; `chart --no-resample` is the one that holds every row.
103
+ `chart` reduces as it reads, and a histogram counts as it reads, so the memory
104
+ they need does not grow with the table; `chart --no-resample` is the one that
105
+ holds every row.
104
106
 
105
107
  ## Queries
106
108
 
@@ -131,15 +133,18 @@ The first column is the X axis (numbers or ISO dates), every other column is a
131
133
  series named after the column; rows with NULLs are skipped and counted on
132
134
  stderr. `--kind` selects `line` (default), `scatter` or `hist` (first column
133
135
  only, `--bins`). `--height`, `--width`, `--x-label`, `--y-label` and `--color`
134
- adjust the drawing.
136
+ adjust the drawing; the Y label is cut to eight characters, the room plotille
137
+ gives it.
135
138
 
136
139
  Line and scatter charts are reduced to what the canvas can show, keeping the
137
140
  minimum and the maximum of every column of braille dots, so spikes survive and
138
- the true extremes keep the X of their own row. The rows are reduced as they are
141
+ the true extremes keep the X of their own row; a scatter plot keeps a uniform
142
+ sample of its points as well, for its density. The rows are reduced as they are
139
143
  read, in chunks, so a chart over millions of rows needs no more memory than one
140
144
  over a thousand; the reduction is reported on stderr. `--no-resample` reads and
141
145
  plots every row instead. Histograms are never reduced, since dropping rows would
142
- change the distribution.
146
+ change the distribution: the query runs twice, once for the range and once for
147
+ the counts, so they do not hold the rows either.
143
148
 
144
149
  ## Export, import, dump and diff
145
150
 
@@ -153,8 +158,10 @@ sqlitexplorer diff app.db backup.db # exit status 1 when th
153
158
 
154
159
  `import` creates the table when it does not exist, inferring INTEGER, REAL or
155
160
  TEXT for each column; values with leading zeros and integers too large for
156
- SQLite stay TEXT, so codes such as `007` keep their digits. Empty CSV cells
157
- become NULL. The format comes from the file extension unless `--format` is
161
+ SQLite stay TEXT, so codes such as `007` keep their digits. From JSON the types
162
+ are the JSON types: numbers become INTEGER or REAL and strings stay TEXT even
163
+ when they look like numbers, so a json export imports back as it was. Empty CSV
164
+ cells become NULL. The format comes from the file extension unless `--format` is
158
165
  given, and `--encoding` reads a file that is not UTF-8.
159
166
 
160
167
  `export --all` writes one file per table and view inside the directory: a name
@@ -4,7 +4,7 @@ build-backend = "setuptools.build_meta"
4
4
 
5
5
  [project]
6
6
  name = "sqlitexplorer"
7
- version = "1.1.3"
7
+ version = "1.2.0"
8
8
  description = "Explore SQLite databases from the terminal: tables, schema, stats, search, queries, charts, export/import and an interactive shell."
9
9
  readme = "README.md"
10
10
  license = "MIT"
@@ -29,7 +29,7 @@ classifiers = [
29
29
  dependencies = [
30
30
  "outfancy>=0.11",
31
31
  "plotille>=6",
32
- "plotilleresample>=1.0",
32
+ "plotilleresample>=1.2",
33
33
  "typer>=0.12",
34
34
  ]
35
35
 
@@ -6,6 +6,7 @@ other column is a numeric series. Histograms use the first column only.
6
6
 
7
7
  from __future__ import annotations
8
8
 
9
+ import functools
9
10
  import os
10
11
  from collections.abc import Callable, Iterator, Sequence
11
12
  from contextlib import contextmanager
@@ -13,6 +14,7 @@ from dataclasses import dataclass
13
14
  from datetime import datetime
14
15
  from enum import Enum
15
16
  from itertools import islice
17
+ from typing import NamedTuple
16
18
 
17
19
  import plotille
18
20
  import plotilleresample
@@ -22,12 +24,13 @@ from sqlitexplorer.render import strip_ansi
22
24
 
23
25
  __all__ = [
24
26
  "ChartKind",
27
+ "Histogram",
25
28
  "Series",
26
- "histogram_values",
27
29
  "render_chart",
28
30
  "render_histogram",
29
31
  "resample_series",
30
32
  "series_from_result",
33
+ "stream_histogram",
31
34
  "stream_series",
32
35
  ]
33
36
 
@@ -39,9 +42,9 @@ AXIS_LABEL_WIDTH = 8
39
42
  AXIS_WIDTH = 13
40
43
  # Narrowest canvas worth drawing on when the labels ask for too much room.
41
44
  MIN_CANVAS = 10
42
- # A tuple, not bool | int | float: the union would be rebuilt on every call,
43
- # and this runs once per value of the result.
44
- _NUMERIC = (bool, int, float)
45
+ # A tuple, not int | float: the union would be rebuilt on every call, and
46
+ # this runs once per value of the result. bool is an int.
47
+ _NUMERIC = (int, float)
45
48
  # Rows read at a time when streaming, and how many reduced points may pile
46
49
  # up before they are reduced again.
47
50
  _CHUNK = 65536
@@ -61,6 +64,15 @@ class Series:
61
64
  y: list[float]
62
65
 
63
66
 
67
+ class Histogram(NamedTuple):
68
+ """Counts of the first column of a query, in the bins plotille would draw."""
69
+
70
+ counts: list[int]
71
+ edges: list[float]
72
+ column: str
73
+ skipped: int
74
+
75
+
64
76
  def _number(value: object) -> float | None:
65
77
  if isinstance(value, _NUMERIC):
66
78
  return float(value)
@@ -131,11 +143,12 @@ def resample_series(
131
143
  """Reduce every series to the points the canvas can actually draw.
132
144
 
133
145
  min/max keeps the extremes of every bucket, so spikes survive, and it only
134
- indexes X, which the LTTB resamplers cannot do when X is a date.
146
+ indexes X, which the LTTB resamplers cannot do when X is a date. A scatter
147
+ plot gets a uniform stride for its density plus those extremes.
135
148
  """
136
149
  budget = _canvas_width(width)
137
150
  reduce = (
138
- plotilleresample.resample_scatter
151
+ plotilleresample.resample_scatter_minmax
139
152
  if kind is ChartKind.SCATTER
140
153
  else plotilleresample.resample_plot_minmax
141
154
  )
@@ -190,24 +203,51 @@ def stream_series(
190
203
  return resample_series(reduced, kind=kind, width=width, height=height), skipped, rows_read
191
204
 
192
205
 
193
- def histogram_values(result: ResultSet) -> tuple[list[float], int]:
194
- """Numeric values of the first column of *result*, and how many NULLs were skipped."""
195
- if not result.columns:
196
- raise ExplorerError("need a numeric column to plot")
197
- name = result.columns[0]
198
- values: list[float] = []
206
+ def stream_histogram(open_stream: Callable[[], RowStream], *, bins: int) -> Histogram:
207
+ """Count the first column of a query into *bins* without holding its values.
208
+
209
+ The query runs twice: the first pass validates the values and finds their
210
+ range, the second counts them, so only the counts stay in memory. The
211
+ bins are the ones plotille computes from the raw values: equal widths
212
+ from the minimum to the maximum, the last one closed. ``skipped`` counts
213
+ the rows whose first column was NULL.
214
+ """
215
+ first = open_stream()
216
+ if not first.returns_rows:
217
+ raise ExplorerError("the statement returned no rows")
218
+ column = first.columns[0]
219
+ low = high = None
199
220
  skipped = 0
200
- for row in result.rows:
201
- if row[0] is None:
202
- skipped += 1
203
- continue
204
- number = _number(row[0])
221
+ for row in first.rows:
222
+ number = _histogram_value(row[0], column)
205
223
  if number is None:
206
- raise ExplorerError(f"column {name} is not numeric: {row[0]!r}")
207
- values.append(number)
208
- if not values:
224
+ skipped += 1
225
+ elif low is None or high is None:
226
+ low = high = number
227
+ else:
228
+ low, high = min(low, number), max(high, number)
229
+ if low is None or high is None:
209
230
  raise ExplorerError("no rows to plot")
210
- return values, skipped
231
+ if low == high:
232
+ low, high = low - 0.5, high + 0.5
233
+ step = (high - low) / bins
234
+ counts = [0] * bins
235
+ for row in open_stream().rows:
236
+ number = _histogram_value(row[0], column)
237
+ if number is not None:
238
+ # Clamped: a query that is not deterministic may not repeat its range.
239
+ counts[max(0, min(bins - 1, int((number - low) // step)))] += 1
240
+ edges = [low + index * step for index in range(bins + 1)]
241
+ return Histogram(counts, edges, column, skipped)
242
+
243
+
244
+ def _histogram_value(value: object, column: str) -> float | None:
245
+ if value is None:
246
+ return None
247
+ number = _number(value)
248
+ if number is None:
249
+ raise ExplorerError(f"column {column} is not numeric: {value!r}")
250
+ return number
211
251
 
212
252
 
213
253
  @contextmanager
@@ -233,19 +273,31 @@ def _canvas_width(width: int) -> int:
233
273
  return max(MIN_CANVAS, width - AXIS_WIDTH)
234
274
 
235
275
 
236
- def _fit(draw: Callable[[int], str], width: int) -> str:
276
+ def _fit(
277
+ draw: Callable[[int], str], width: int, *, probe: Callable[[int], str] | None = None
278
+ ) -> str:
237
279
  """Draw on the widest canvas whose longest line still fits in *width*.
238
280
 
239
281
  plotille writes the X label and the tick numbers past the end of the
240
282
  canvas, by an amount that depends on both, so the only way to know the
241
- room they take is to draw and measure.
283
+ room they take is to draw and measure. That room does not depend on the
284
+ number of points, so *probe*, a drawing with the same labels and the same
285
+ ranges but two points per series, finds the canvas cheaply and *draw*
286
+ then runs once.
242
287
  """
243
288
  canvas = _canvas_width(width)
289
+ if probe is not None:
290
+ canvas = _narrow(probe, width, canvas)[1]
291
+ return _narrow(draw, width, canvas)[0]
292
+
293
+
294
+ def _narrow(draw: Callable[[int], str], width: int, canvas: int) -> tuple[str, int]:
295
+ """Shrink *canvas* until the drawing fits in *width*; return the drawing and its canvas."""
244
296
  while True:
245
297
  drawing = draw(canvas)
246
298
  excess = max(len(strip_ansi(line)) for line in drawing.splitlines()) - width
247
299
  if excess <= 0 or canvas <= MIN_CANVAS:
248
- return drawing
300
+ return drawing, canvas
249
301
  canvas = max(MIN_CANVAS, canvas - excess)
250
302
 
251
303
 
@@ -260,50 +312,81 @@ def render_chart(
260
312
  y_label: str,
261
313
  ) -> str:
262
314
  """Draw *series* as a line chart or scatter plot, with a legend when there are several."""
315
+ draw = functools.partial(
316
+ _draw_series, kind=kind, height=height, color=color, x_label=x_label, y_label=y_label
317
+ )
318
+ return _fit(
319
+ lambda canvas: draw(series, canvas=canvas),
320
+ width,
321
+ probe=lambda canvas: draw(_extremes(series), canvas=canvas),
322
+ )
263
323
 
264
- def draw(canvas: int) -> str:
265
- figure = plotille.Figure()
266
- figure.width = canvas
267
- figure.height = max(3, height)
268
- figure.with_colors = color
269
- figure.color_mode = "names"
270
- figure.x_label = x_label
271
- figure.y_label = y_label[:AXIS_LABEL_WIDTH]
272
- for index, item in enumerate(series):
273
- line_color = PALETTE[index % len(PALETTE)] if color else None
274
- if kind is ChartKind.SCATTER:
275
- figure.scatter(item.x, item.y, lc=line_color, label=item.label)
276
- else:
277
- figure.plot(item.x, item.y, lc=line_color, label=item.label)
278
- with _color_environment(color):
279
- return figure.show(legend=len(series) > 1)
280
324
 
281
- return _fit(draw, width)
325
+ def _figure(
326
+ *, canvas: int, height: int, color: bool, x_label: str, y_label: str
327
+ ) -> plotille.Figure:
328
+ figure = plotille.Figure()
329
+ figure.width = canvas
330
+ figure.height = max(3, height)
331
+ figure.with_colors = color
332
+ figure.color_mode = "names"
333
+ figure.x_label = x_label
334
+ figure.y_label = y_label[:AXIS_LABEL_WIDTH]
335
+ return figure
282
336
 
283
337
 
284
- def render_histogram(
285
- values: Sequence[float],
338
+ def _draw_series(
339
+ series: Sequence[Series],
286
340
  *,
287
- bins: int,
288
- width: int,
341
+ kind: ChartKind,
342
+ canvas: int,
289
343
  height: int,
290
344
  color: bool,
291
345
  x_label: str,
292
346
  y_label: str,
293
347
  ) -> str:
294
- """Draw the distribution of *values*."""
295
- numbers = list(values)
296
-
297
- def draw(canvas: int) -> str:
298
- with _color_environment(color):
299
- return plotille.histogram(
300
- numbers,
301
- bins=bins,
302
- width=canvas,
303
- height=max(3, height),
304
- X_label=x_label,
305
- Y_label=y_label[:AXIS_LABEL_WIDTH],
306
- lc=PALETTE[0] if color else None,
307
- )
308
-
309
- return _fit(draw, width)
348
+ figure = _figure(canvas=canvas, height=height, color=color, x_label=x_label, y_label=y_label)
349
+ for index, item in enumerate(series):
350
+ line_color = PALETTE[index % len(PALETTE)] if color else None
351
+ if kind is ChartKind.SCATTER:
352
+ figure.scatter(item.x, item.y, lc=line_color, label=item.label)
353
+ else:
354
+ figure.plot(item.x, item.y, lc=line_color, label=item.label)
355
+ with _color_environment(color):
356
+ return figure.show(legend=len(series) > 1)
357
+
358
+
359
+ def _extremes(series: Sequence[Series]) -> list[Series]:
360
+ """Two points per series, on the same ranges: enough to lay the axes out."""
361
+ return [
362
+ Series(label=item.label, x=[min(item.x), max(item.x)], y=[min(item.y), max(item.y)])
363
+ for item in series
364
+ ]
365
+
366
+
367
+ def render_histogram(
368
+ histogram: Histogram, *, width: int, height: int, color: bool, x_label: str, y_label: str
369
+ ) -> str:
370
+ """Draw *histogram* the way plotille draws one from the raw values."""
371
+ draw = functools.partial(
372
+ _draw_histogram, histogram, height=height, color=color, x_label=x_label, y_label=y_label
373
+ )
374
+ return _fit(lambda canvas: draw(canvas=canvas), width)
375
+
376
+
377
+ def _draw_histogram(
378
+ histogram: Histogram, *, canvas: int, height: int, color: bool, x_label: str, y_label: str
379
+ ) -> str:
380
+ figure = _figure(canvas=canvas, height=height, color=color, x_label=x_label, y_label=y_label)
381
+ # plotille only bins raw values, but its Histogram plot keeps the counts
382
+ # apart from them and draws from the counts alone: the two edges give it
383
+ # the same range, hence the same bins, and the real counts then replace
384
+ # the ones it took from those two values.
385
+ figure.histogram(
386
+ [histogram.edges[0], histogram.edges[-1]],
387
+ bins=len(histogram.counts),
388
+ lc=PALETTE[0] if color else None,
389
+ )
390
+ figure._plots[-1].frequencies = list(histogram.counts)
391
+ with _color_environment(color):
392
+ return figure.show()
@@ -25,10 +25,10 @@ import typer
25
25
  from sqlitexplorer import __version__
26
26
  from sqlitexplorer.charts import (
27
27
  ChartKind,
28
- histogram_values,
29
28
  render_chart,
30
29
  render_histogram,
31
30
  series_from_result,
31
+ stream_histogram,
32
32
  stream_series,
33
33
  )
34
34
  from sqlitexplorer.completion import complete_table
@@ -203,9 +203,15 @@ def _fail(message: str) -> NoReturn:
203
203
 
204
204
  @contextmanager
205
205
  def _reporting_errors() -> Iterator[None]:
206
- """Turn ExplorerError and OSError into a message on stderr and exit status 1."""
206
+ """Turn ExplorerError and OSError into a message on stderr and exit status 1.
207
+
208
+ A closed pipe (``| head`` stopped reading) is not an error to report: it
209
+ is left to Typer, which ends the command quietly.
210
+ """
207
211
  try:
208
212
  yield
213
+ except BrokenPipeError:
214
+ raise
209
215
  except ExplorerError as error:
210
216
  _fail(_error_message(error))
211
217
  except OSError as error:
@@ -657,7 +663,12 @@ def chart(
657
663
  ),
658
664
  ] = True,
659
665
  x_label: Annotated[str | None, typer.Option("--x-label", help="Label of the X axis.")] = None,
660
- y_label: Annotated[str | None, typer.Option("--y-label", help="Label of the Y axis.")] = None,
666
+ y_label: Annotated[
667
+ str | None,
668
+ typer.Option(
669
+ "--y-label", help="Label of the Y axis, at most 8 characters (plotille's limit)."
670
+ ),
671
+ ] = None,
661
672
  params: ParamOption = None,
662
673
  width: WidthOption = None,
663
674
  color: ColorOption = None,
@@ -670,17 +681,16 @@ def chart(
670
681
  use_color = resolve_color(color)
671
682
  screen = width if width is not None else shutil.get_terminal_size().columns
672
683
  if kind is ChartKind.HIST:
673
- result = db.execute(text, bound)
674
- if not result.returns_rows:
675
- _fail("the statement returned no rows")
676
- values, skipped = histogram_values(result)
684
+ # Two passes over the query, the range first and then the counts,
685
+ # so the values never pile up in memory.
686
+ histogram = stream_histogram(lambda: db.stream(text, bound), bins=bins)
687
+ skipped = histogram.skipped
677
688
  drawing = render_histogram(
678
- values,
679
- bins=bins,
689
+ histogram,
680
690
  width=screen,
681
691
  height=height,
682
692
  color=use_color,
683
- x_label=x_label or result.columns[0],
693
+ x_label=x_label or histogram.column,
684
694
  y_label=y_label or "count",
685
695
  )
686
696
  else:
@@ -693,8 +703,9 @@ def chart(
693
703
  series, skipped, read = stream_series(
694
704
  stream, kind=kind, width=screen, height=height
695
705
  )
696
- if len(series[0].x) < read:
697
- typer.echo(f"resampled {read} rows to {len(series[0].x)} points", err=True)
706
+ plotted, points = read - skipped, len(series[0].x)
707
+ if points < plotted:
708
+ typer.echo(f"resampled {plotted} rows to {points} points", err=True)
698
709
  else:
699
710
  result = db.execute(text, bound)
700
711
  if not result.returns_rows:
@@ -831,7 +842,10 @@ def import_(
831
842
  text = _read_text(file, encoding)
832
843
  with _reporting_errors(), open_database(database, write=True) as db:
833
844
  headers, raw_rows = parse_rows(text, input_format, delimiter=delimiter)
834
- types = infer_types(raw_rows, len(headers))
845
+ # A JSON file brings its own types; a CSV only has text to go by.
846
+ types = infer_types(
847
+ raw_rows, len(headers), parse_text=input_format is not OutputFormat.JSON
848
+ )
835
849
  count = db.import_rows(
836
850
  table, list(zip(headers, types, strict=True)), coerce_rows(raw_rows, types)
837
851
  )
@@ -8,9 +8,10 @@ that should reach the user as a plain message is raised as
8
8
 
9
9
  from __future__ import annotations
10
10
 
11
+ import re
11
12
  import sqlite3
12
13
  from collections.abc import Iterable, Iterator, Mapping, Sequence
13
- from contextlib import contextmanager
14
+ from contextlib import contextmanager, suppress
14
15
  from dataclasses import dataclass, field
15
16
  from itertools import islice
16
17
  from pathlib import Path
@@ -100,10 +101,16 @@ class StatsReport(NamedTuple):
100
101
 
101
102
 
102
103
  def _rows_of(cursor: sqlite3.Cursor) -> Iterator[tuple]:
104
+ # Not ``yield from``: closing the generator would then close the cursor
105
+ # itself, outside the ``finally`` below.
103
106
  try:
104
- yield from cursor
107
+ for row in cursor: # noqa: UP028
108
+ yield row
105
109
  finally:
106
- cursor.close()
110
+ # A stream cut short by an error, Ctrl-C or a closed pipe is finalised
111
+ # after the connection is closed, which already released the cursor.
112
+ with suppress(sqlite3.ProgrammingError):
113
+ cursor.close()
107
114
 
108
115
 
109
116
  def _window(stream: RowStream, page: Page) -> ResultSet:
@@ -129,7 +136,8 @@ def split_statements(sql: str) -> list[str]:
129
136
 
130
137
  Uses :func:`sqlite3.complete_statement`, so semicolons inside strings,
131
138
  comments and ``CREATE TRIGGER ... END`` blocks do not split. Trailing text
132
- without a semicolon is returned as a final statement.
139
+ without a semicolon is returned as a final statement; text that holds
140
+ only whitespace, semicolons and comments is not a statement.
133
141
  """
134
142
  statements: list[str] = []
135
143
  buffer = ""
@@ -147,7 +155,8 @@ def split_statements(sql: str) -> list[str]:
147
155
 
148
156
 
149
157
  def _has_content(text: str) -> bool:
150
- return bool(text.strip().rstrip(";").strip())
158
+ """Whether *text* holds anything besides whitespace, semicolons and comments."""
159
+ return bool(_COMMENT.sub("", text).strip().rstrip(";").strip())
151
160
 
152
161
 
153
162
  def plan_tree(result: ResultSet) -> ResultSet:
@@ -695,6 +704,8 @@ class Explorer:
695
704
  cursor.close()
696
705
 
697
706
 
707
+ # SQL comments, removed only to decide whether a statement is empty.
708
+ _COMMENT = re.compile(r"--[^\n]*|/\*.*?\*/", re.DOTALL)
698
709
  _SEARCH_COLUMNS = ("table", "column", "rowid", "value")
699
710
  _STATS_COLUMNS = ("column", "type", "nulls", "distinct", "min", "max", "top")
700
711
  # Columns per aggregate query: five expressions each, well below SQLITE_MAX_COLUMN.
@@ -51,11 +51,11 @@ __all__ = [
51
51
 
52
52
  ELLIPSIS = "…"
53
53
  _ANSI = re.compile(r"\x1b\[[0-9;]*m")
54
- _INTEGER = re.compile(r"^[+-]?\d+$")
55
- _REAL = re.compile(r"^[+-]?(\d+\.\d*|\.\d+|\d+)([eE][+-]?\d+)?$")
56
- _LEADING_ZERO = re.compile(r"^[+-]?0\d")
57
- # A tuple, not bool | int: the union would be rebuilt on every call.
58
- _INTEGRAL = (bool, int)
54
+ # ASCII digits only: \d also matches the digits of other scripts, which
55
+ # int() would happily convert.
56
+ _INTEGER = re.compile(r"^[+-]?[0-9]+$")
57
+ _REAL = re.compile(r"^[+-]?([0-9]+\.[0-9]*|\.[0-9]+|[0-9]+)([eE][+-]?[0-9]+)?$")
58
+ _LEADING_ZERO = re.compile(r"^[+-]?0[0-9]")
59
59
  # Integers outside this range do not fit in a SQLite INTEGER column.
60
60
  _INT64 = range(-(2**63), 2**63)
61
61
 
@@ -195,7 +195,9 @@ def render_table(
195
195
  columns = list(result.columns)
196
196
 
197
197
  screen = width if width is not None else shutil.get_terminal_size().columns
198
- # outfancy keeps a two-column margin and one separator per column.
198
+ # One separator per column, plus two columns of margin: outfancy needs
199
+ # none when the widths are given, but its own corrector leaves the same
200
+ # two, and a console that wraps on the last column would add blank lines.
199
201
  available = screen - 2 - len(columns)
200
202
  widths = fit_widths(columns, _natural_widths(columns, rows), available)
201
203
  return table.render(data=rows, label_list=columns, width=widths, screen_x=width)
@@ -480,11 +482,13 @@ def _plain_json_value(value: object) -> object:
480
482
  return value
481
483
 
482
484
 
483
- def _classify(value: object) -> str:
484
- if isinstance(value, _INTEGRAL):
485
+ def _classify(value: object, *, parse_text: bool) -> str:
486
+ if isinstance(value, int): # bool is an int
485
487
  return "INTEGER"
486
488
  if isinstance(value, float):
487
489
  return "REAL"
490
+ if not parse_text:
491
+ return "TEXT"
488
492
  text = str(value).strip()
489
493
  if _LEADING_ZERO.match(text):
490
494
  # 007 is a code, not a number: storing it as one would drop the zeros.
@@ -496,8 +500,15 @@ def _classify(value: object) -> str:
496
500
  return "TEXT"
497
501
 
498
502
 
499
- def infer_types(rows: Sequence[Sequence[object]], count: int) -> list[str]:
500
- """Pick INTEGER, REAL or TEXT for each of the *count* columns from the values seen."""
503
+ def infer_types(
504
+ rows: Sequence[Sequence[object]], count: int, *, parse_text: bool = True
505
+ ) -> list[str]:
506
+ """Pick INTEGER, REAL or TEXT for each of the *count* columns from the values seen.
507
+
508
+ With *parse_text*, textual values that look like numbers count as numbers,
509
+ which is what a CSV needs. Without it they stay TEXT: a JSON string is a
510
+ string whatever it holds, and its numbers already come as numbers.
511
+ """
501
512
  rank = {"INTEGER": 0, "REAL": 1, "TEXT": 2}
502
513
  types = ["INTEGER"] * count
503
514
  seen = [False] * count
@@ -507,7 +518,7 @@ def infer_types(rows: Sequence[Sequence[object]], count: int) -> list[str]:
507
518
  if value is None:
508
519
  continue
509
520
  seen[index] = True
510
- kind = _classify(value)
521
+ kind = _classify(value, parse_text=parse_text)
511
522
  if rank[kind] > rank[types[index]]:
512
523
  types[index] = kind
513
524
  return [kind if was_seen else "TEXT" for kind, was_seen in zip(types, seen, strict=True)]
@@ -1,6 +1,6 @@
1
1
  Metadata-Version: 2.4
2
2
  Name: sqlitexplorer
3
- Version: 1.1.3
3
+ Version: 1.2.0
4
4
  Summary: Explore SQLite databases from the terminal: tables, schema, stats, search, queries, charts, export/import and an interactive shell.
5
5
  Author-email: "Carlos A. Planchón" <carlosandresplanchonprestes@gmail.com>
6
6
  License-Expression: MIT
@@ -21,7 +21,7 @@ Description-Content-Type: text/markdown
21
21
  License-File: LICENSE
22
22
  Requires-Dist: outfancy>=0.11
23
23
  Requires-Dist: plotille>=6
24
- Requires-Dist: plotilleresample>=1.0
24
+ Requires-Dist: plotilleresample>=1.2
25
25
  Requires-Dist: typer>=0.12
26
26
  Provides-Extra: dev
27
27
  Requires-Dist: pytest>=8; extra == "dev"
@@ -50,10 +50,11 @@ with [plotille](https://github.com/tammoippen/plotille), resampled by
50
50
  Requires Python 3.10 or newer.
51
51
 
52
52
  ```sh
53
- # From a clone of this repository:
54
- uv tool install .
53
+ uv tool install sqlitexplorer
55
54
  # or with pip:
56
- pip install .
55
+ pip install sqlitexplorer
56
+ # or from a clone of this repository:
57
+ uv tool install .
57
58
  ```
58
59
 
59
60
  ## Quick tour
@@ -129,8 +130,9 @@ Big results do not need to fit in memory: `--page` fetches only the requested
129
130
  page (in SQL for `show`, from the cursor for `query`), and the csv, tsv, json
130
131
  and markdown formats, `export` and `dump` are written row by row. The table
131
132
  format is the exception, since it needs every row to size its columns.
132
- `chart` reduces as it reads, so the memory it needs does not grow with the
133
- table; `chart --no-resample` is the one that holds every row.
133
+ `chart` reduces as it reads, and a histogram counts as it reads, so the memory
134
+ they need does not grow with the table; `chart --no-resample` is the one that
135
+ holds every row.
134
136
 
135
137
  ## Queries
136
138
 
@@ -161,15 +163,18 @@ The first column is the X axis (numbers or ISO dates), every other column is a
161
163
  series named after the column; rows with NULLs are skipped and counted on
162
164
  stderr. `--kind` selects `line` (default), `scatter` or `hist` (first column
163
165
  only, `--bins`). `--height`, `--width`, `--x-label`, `--y-label` and `--color`
164
- adjust the drawing.
166
+ adjust the drawing; the Y label is cut to eight characters, the room plotille
167
+ gives it.
165
168
 
166
169
  Line and scatter charts are reduced to what the canvas can show, keeping the
167
170
  minimum and the maximum of every column of braille dots, so spikes survive and
168
- the true extremes keep the X of their own row. The rows are reduced as they are
171
+ the true extremes keep the X of their own row; a scatter plot keeps a uniform
172
+ sample of its points as well, for its density. The rows are reduced as they are
169
173
  read, in chunks, so a chart over millions of rows needs no more memory than one
170
174
  over a thousand; the reduction is reported on stderr. `--no-resample` reads and
171
175
  plots every row instead. Histograms are never reduced, since dropping rows would
172
- change the distribution.
176
+ change the distribution: the query runs twice, once for the range and once for
177
+ the counts, so they do not hold the rows either.
173
178
 
174
179
  ## Export, import, dump and diff
175
180
 
@@ -183,8 +188,10 @@ sqlitexplorer diff app.db backup.db # exit status 1 when th
183
188
 
184
189
  `import` creates the table when it does not exist, inferring INTEGER, REAL or
185
190
  TEXT for each column; values with leading zeros and integers too large for
186
- SQLite stay TEXT, so codes such as `007` keep their digits. Empty CSV cells
187
- become NULL. The format comes from the file extension unless `--format` is
191
+ SQLite stay TEXT, so codes such as `007` keep their digits. From JSON the types
192
+ are the JSON types: numbers become INTEGER or REAL and strings stay TEXT even
193
+ when they look like numbers, so a json export imports back as it was. Empty CSV
194
+ cells become NULL. The format comes from the file extension unless `--format` is
188
195
  given, and `--encoding` reads a file that is not UTF-8.
189
196
 
190
197
  `export --all` writes one file per table and view inside the directory: a name
@@ -1,6 +1,6 @@
1
1
  outfancy>=0.11
2
2
  plotille>=6
3
- plotilleresample>=1.0
3
+ plotilleresample>=1.2
4
4
  typer>=0.12
5
5
 
6
6
  [dev]
@@ -4,18 +4,21 @@ from __future__ import annotations
4
4
 
5
5
  from datetime import datetime
6
6
 
7
+ import plotille
7
8
  import pytest
8
9
 
9
10
  from sqlitexplorer.charts import (
10
11
  AXIS_WIDTH,
11
12
  MIN_CANVAS,
12
13
  ChartKind,
14
+ Histogram,
13
15
  Series,
14
- histogram_values,
16
+ _draw_histogram,
15
17
  render_chart,
16
18
  render_histogram,
17
19
  resample_series,
18
20
  series_from_result,
21
+ stream_histogram,
19
22
  stream_series,
20
23
  )
21
24
  from sqlitexplorer.core import ExplorerError, ResultSet, RowStream
@@ -67,20 +70,6 @@ def test_series_requires_two_columns() -> None:
67
70
  series_from_result(ResultSet(columns=("x",), rows=[(1,)]))
68
71
 
69
72
 
70
- def test_histogram_values_uses_first_column() -> None:
71
- result = ResultSet(columns=("v", "other"), rows=[(1, "a"), (None, "b"), (2.5, "c")])
72
- values, skipped = histogram_values(result)
73
- assert values == [1.0, 2.5]
74
- assert skipped == 1
75
- with pytest.raises(ExplorerError, match="not numeric"):
76
- histogram_values(ResultSet(columns=("v",), rows=[("x",)]))
77
-
78
-
79
- def test_histogram_values_requires_a_column() -> None:
80
- with pytest.raises(ExplorerError, match="need a numeric column"):
81
- histogram_values(ResultSet())
82
-
83
-
84
73
  def test_render_chart_prints_braille_without_colors() -> None:
85
74
  series, _ = series_from_result(ResultSet(columns=("x", "y"), rows=[(1, 1), (2, 3), (3, 2)]))
86
75
  text = render_chart(
@@ -110,15 +99,6 @@ def test_render_chart_single_point_does_not_crash() -> None:
110
99
  assert braille(text)
111
100
 
112
101
 
113
- def test_render_histogram() -> None:
114
- text = render_histogram(
115
- [1.0, 2.0, 2.0, 3.0], bins=3, width=50, height=6, color=False, x_label="v", y_label="count"
116
- )
117
- assert braille(text)
118
- assert "(v)" in text
119
- assert "\x1b[" not in text
120
-
121
-
122
102
  def test_resample_series_reduces_and_keeps_the_extremes() -> None:
123
103
  rows: list[tuple] = [(i, 0.0) for i in range(5000)]
124
104
  rows[1234] = (1234, 999.0)
@@ -163,6 +143,68 @@ def _stream(columns: tuple[str, ...], rows: list[tuple]) -> RowStream:
163
143
  return RowStream(columns=columns, rows=iter(rows))
164
144
 
165
145
 
146
+ def _histogram(values: list[object], bins: int = 3) -> Histogram:
147
+ return stream_histogram(lambda: _stream(("v", "other"), [(v, "x") for v in values]), bins=bins)
148
+
149
+
150
+ def test_stream_histogram_counts_the_first_column_in_two_passes() -> None:
151
+ opened = 0
152
+
153
+ def open_stream() -> RowStream:
154
+ nonlocal opened
155
+ opened += 1
156
+ return _stream(("v", "other"), [(1, "a"), (None, "b"), (2.5, "c"), (4, "d")])
157
+
158
+ histogram = stream_histogram(open_stream, bins=3)
159
+ assert opened == 2
160
+ assert histogram.column == "v"
161
+ assert histogram.skipped == 1
162
+ assert histogram.edges == [1.0, 2.0, 3.0, 4.0]
163
+ assert histogram.counts == [1, 1, 1] # the last bin is closed: 4 lands in it
164
+
165
+
166
+ def test_stream_histogram_single_value_gets_a_unit_range() -> None:
167
+ histogram = _histogram([7, 7, 7], bins=2)
168
+ assert histogram.edges == [6.5, 7.0, 7.5]
169
+ assert histogram.counts == [0, 3]
170
+
171
+
172
+ def test_stream_histogram_rejects_bad_input() -> None:
173
+ with pytest.raises(ExplorerError, match="not numeric"):
174
+ _histogram(["x"])
175
+ with pytest.raises(ExplorerError, match="no rows to plot"):
176
+ _histogram([None, None])
177
+ with pytest.raises(ExplorerError, match="returned no rows"):
178
+ stream_histogram(lambda: RowStream(), bins=3)
179
+
180
+
181
+ def test_render_histogram() -> None:
182
+ histogram = _histogram([1.0, 2.0, 2.0, 3.0])
183
+ text = render_histogram(
184
+ histogram, width=50, height=6, color=False, x_label="v", y_label="count"
185
+ )
186
+ assert braille(text)
187
+ assert "(v)" in text
188
+ assert "\x1b[" not in text
189
+ colored = render_histogram(
190
+ histogram, width=50, height=6, color=True, x_label="v", y_label="count"
191
+ )
192
+ assert "\x1b[" in colored
193
+
194
+
195
+ def test_render_histogram_matches_plotille(monkeypatch: pytest.MonkeyPatch) -> None:
196
+ # The counts are fed to plotille's own Histogram plot, so a plotille
197
+ # release that changes how it keeps them would show up here.
198
+ monkeypatch.setenv("NO_COLOR", "1")
199
+ values = [float(i % 97) for i in range(2000)] + [3.5, 3.5, 96.0]
200
+ ours = _draw_histogram(
201
+ _histogram(values, bins=12), canvas=50, height=8, color=False, x_label="v", y_label="count"
202
+ )
203
+ theirs = plotille.histogram(values, bins=12, width=50, height=8, X_label="v", Y_label="count")
204
+ assert ours == theirs
205
+ assert braille(ours)
206
+
207
+
166
208
  def test_stream_series_keeps_the_envelope_and_the_real_pairs(
167
209
  monkeypatch: pytest.MonkeyPatch,
168
210
  ) -> None:
@@ -262,8 +304,7 @@ def test_render_chart_never_draws_wider_than_the_terminal(width: int, x_label: s
262
304
  @pytest.mark.parametrize("width", [40, 60, 80, 96, 120, 200])
263
305
  def test_render_histogram_never_draws_wider_than_the_terminal(width: int) -> None:
264
306
  drawing = render_histogram(
265
- [float(i % 97) for i in range(2000)],
266
- bins=10,
307
+ _histogram([float(i % 97) for i in range(2000)], bins=10),
267
308
  width=width,
268
309
  height=6,
269
310
  color=False,
@@ -289,3 +330,37 @@ def test_render_chart_stops_narrowing_at_the_floor() -> None:
289
330
  body = [line for line in drawing.splitlines() if braille(line)]
290
331
  assert body, drawing
291
332
  assert max(len(line) for line in body) == MIN_CANVAS + AXIS_WIDTH
333
+
334
+
335
+ def test_render_chart_draws_the_full_series_once(monkeypatch: pytest.MonkeyPatch) -> None:
336
+ # The canvas is found on a two-point stand-in; the real points are drawn once.
337
+ lengths: list[int] = []
338
+ original = plotille.Figure.plot
339
+
340
+ def recording(self: plotille.Figure, X: list, Y: list, **kwargs: object) -> None:
341
+ lengths.append(len(X))
342
+ original(self, X, Y, **kwargs)
343
+
344
+ monkeypatch.setattr(plotille.Figure, "plot", recording)
345
+ xs = [float(i) for i in range(1000)]
346
+ render_chart(
347
+ [Series(label="v", x=xs, y=[i % 7 for i in range(1000)])],
348
+ kind=ChartKind.LINE,
349
+ width=80,
350
+ height=9,
351
+ color=False,
352
+ x_label="t",
353
+ y_label="v",
354
+ )
355
+ assert lengths.count(1000) == 1
356
+ assert set(lengths) == {2, 1000}
357
+
358
+
359
+ def test_resample_series_scatter_keeps_an_isolated_spike() -> None:
360
+ rows: list[tuple] = [(i, 0.0) for i in range(50_000)]
361
+ rows[12_347] = (12_347, 999.0)
362
+ series, _ = series_from_result(ResultSet(columns=("x", "y"), rows=rows))
363
+ reduced = resample_series(series, kind=ChartKind.SCATTER, width=80, height=15)
364
+ assert len(reduced[0].x) < 50_000
365
+ assert 999.0 in reduced[0].y
366
+ assert reduced[0].x[reduced[0].y.index(999.0)] == 12_347
@@ -3,14 +3,18 @@
3
3
  from __future__ import annotations
4
4
 
5
5
  import json
6
+ import os
6
7
  import re
7
8
  import sqlite3
9
+ import sys
8
10
  from collections.abc import Callable
11
+ from contextlib import suppress
9
12
  from pathlib import Path
10
13
 
11
14
  import pytest
12
15
 
13
16
  from sqlitexplorer import __version__
17
+ from sqlitexplorer.cli import main
14
18
 
15
19
  # --- Basics -------------------------------------------------------------------
16
20
 
@@ -35,6 +39,26 @@ def test_missing_database_is_a_usage_error(invoke: Callable, tmp_path: Path) ->
35
39
  assert not missing.exists()
36
40
 
37
41
 
42
+ def test_broken_pipe_exits_quietly(
43
+ database: Path, monkeypatch: pytest.MonkeyPatch, capsys: pytest.CaptureFixture[str]
44
+ ) -> None:
45
+ # ``sqlitexplorer show ... | head``: the reader goes away, the first
46
+ # write fails with EPIPE and the command must end without a word.
47
+ read_end, write_end = os.pipe()
48
+ os.close(read_end)
49
+ sink = os.fdopen(write_end, "w", encoding="utf-8")
50
+ monkeypatch.setattr(sys, "stdout", sink)
51
+ monkeypatch.setattr(sys, "argv", ["sqlitexplorer", "show", str(database), "users", "-f", "csv"])
52
+ try:
53
+ with pytest.raises(SystemExit) as exit_info:
54
+ main()
55
+ finally:
56
+ with suppress(OSError):
57
+ sink.close()
58
+ assert exit_info.value.code == 1
59
+ assert capsys.readouterr().err == ""
60
+
61
+
38
62
  # --- tables / schema / describe ----------------------------------------------
39
63
 
40
64
 
@@ -463,6 +487,7 @@ def test_query_file_runs_every_statement(
463
487
  "INSERT INTO users (name) VALUES ('Zoe');\n"
464
488
  "-- a comment; with a semicolon\n"
465
489
  "SELECT COUNT(*) AS total FROM users;\n"
490
+ "-- a trailing comment is not a statement\n"
466
491
  )
467
492
  result = invoke("query", database, "--file", script, "--write", "-f", "csv")
468
493
  assert result.exit_code == 0
@@ -568,6 +593,7 @@ def test_chart_line_prints_braille(invoke: Callable, database: Path) -> None:
568
593
  assert braille(result.stdout)
569
594
  assert "\x1b[" not in result.stdout
570
595
  assert "skipped 1 row with NULL values" in result.stderr
596
+ assert "resampled" not in result.stderr # two rows do not need reducing
571
597
 
572
598
 
573
599
  def test_chart_hist_uses_first_column(invoke: Callable, database: Path) -> None:
@@ -678,6 +704,19 @@ def test_import_into_existing_table_and_json(
678
704
  ]
679
705
 
680
706
 
707
+ def test_import_json_keeps_strings_as_text(
708
+ invoke: Callable, database: Path, tmp_path: Path
709
+ ) -> None:
710
+ source = tmp_path / "typed.json"
711
+ source.write_text('[{"zip": "12345", "n": 7}, {"zip": "67890", "n": 8}]')
712
+ result = invoke("import", database, "typed", source)
713
+ assert result.exit_code == 0
714
+ described = invoke("describe", database, "typed", "-f", "csv").output.splitlines()
715
+ assert [line.split(",")[2] for line in described[1:]] == ["TEXT", "INTEGER"]
716
+ rows = invoke("query", database, "SELECT typeof(zip), zip FROM typed ORDER BY zip", "-f", "csv")
717
+ assert rows.output.splitlines() == ["typeof(zip),zip", "text,12345", "text,67890"]
718
+
719
+
681
720
  def test_import_rolls_back_on_error(invoke: Callable, database: Path, tmp_path: Path) -> None:
682
721
  source = tmp_path / "bad.csv"
683
722
  source.write_text("x,y\n1,2\n")
@@ -830,13 +869,16 @@ def test_chart_resamples_lines_by_default_but_never_histograms(
830
869
  connection.executemany(
831
870
  "INSERT INTO m VALUES (?, ?)", [(i, float(i % 100)) for i in range(5000)]
832
871
  )
872
+ connection.executemany("INSERT INTO m VALUES (?, NULL)", [(i,) for i in range(5000, 5010)])
833
873
  connection.commit()
834
874
  finally:
835
875
  connection.close()
836
876
  args = ("chart", source, "SELECT t, v FROM m", "--width", "80", "--no-color")
837
877
  result = invoke(*args)
838
878
  assert result.exit_code == 0
879
+ # Rows skipped for their NULLs are reported apart, not as resampled.
839
880
  assert "resampled 5000 rows to" in result.stderr
881
+ assert "skipped 10 rows with NULL values" in result.stderr
840
882
  plain = invoke(*args, "--no-resample")
841
883
  assert plain.exit_code == 0
842
884
  assert "resampled" not in plain.stderr
@@ -49,6 +49,10 @@ def test_split_statements_handles_strings_comments_and_triggers() -> None:
49
49
  ]
50
50
  assert split_statements(" ;; \n") == []
51
51
  assert split_statements("SELECT 1;") == ["SELECT 1;"]
52
+ # Comments after the last statement are not a statement of their own.
53
+ assert split_statements("SELECT 1; -- done") == ["SELECT 1;"]
54
+ assert split_statements("/* nothing */ ; -- at all\n") == []
55
+ assert split_statements("SELECT '--'; SELECT '/*'") == ["SELECT '--';", "SELECT '/*'"]
52
56
 
53
57
 
54
58
  def test_plan_tree_indents_by_depth() -> None:
@@ -155,6 +159,15 @@ def test_stream_yields_rows_lazily(database: Path) -> None:
155
159
  assert streamed.rows == db.rows("users", columns=["name"], order_by="id", limit=2).rows
156
160
 
157
161
 
162
+ def test_abandoned_stream_is_quiet_after_the_connection_closed(database: Path) -> None:
163
+ # A stream cut short by an error, Ctrl-C or a closed pipe is finalised
164
+ # once the connection is gone; the cursor is already released by then.
165
+ with open_database(database) as db:
166
+ stream = db.stream("SELECT id FROM users ORDER BY id")
167
+ assert next(stream.rows) == (1,)
168
+ stream.rows.close()
169
+
170
+
158
171
  def test_explain_returns_an_indented_plan(database: Path) -> None:
159
172
  with open_database(database) as db:
160
173
  result = db.explain("SELECT * FROM users WHERE name = 'Marie'")
@@ -167,6 +167,7 @@ def test_parse_number() -> None:
167
167
  assert parse_number("007") == 7
168
168
  assert parse_number("abc") == "abc"
169
169
  assert parse_number("1_000") == "1_000"
170
+ assert parse_number("١٢٣") == "١٢٣" # digits of other scripts are not numbers
170
171
 
171
172
 
172
173
  def test_parse_rows_csv_and_json() -> None:
@@ -200,6 +201,14 @@ def test_infer_types_and_coerce_rows() -> None:
200
201
  types = infer_types(rows, 4)
201
202
  assert types == ["INTEGER", "REAL", "TEXT", "TEXT"]
202
203
  assert coerce_rows(rows, types) == [[1, 1.5, "x", None], [2, 2.0, "3", None]]
204
+ # JSON brings its own types: a string stays TEXT whatever it looks like.
205
+ assert infer_types([["1", 2.5, True, None]], 4, parse_text=False) == [
206
+ "TEXT",
207
+ "REAL",
208
+ "INTEGER",
209
+ "TEXT",
210
+ ]
211
+ assert infer_types([["١٢٣"]], 1) == ["TEXT"]
203
212
 
204
213
 
205
214
  def test_infer_types_keeps_codes_and_huge_integers_as_text() -> None:
File without changes
File without changes
File without changes