sqlitexplorer 1.1.1__tar.gz → 1.1.2__tar.gz

This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
Files changed (27) hide show
  1. {sqlitexplorer-1.1.1/sqlitexplorer.egg-info → sqlitexplorer-1.1.2}/PKG-INFO +11 -9
  2. {sqlitexplorer-1.1.1 → sqlitexplorer-1.1.2}/README.md +10 -8
  3. {sqlitexplorer-1.1.1 → sqlitexplorer-1.1.2}/pyproject.toml +1 -1
  4. {sqlitexplorer-1.1.1 → sqlitexplorer-1.1.2}/sqlitexplorer/charts.py +56 -4
  5. {sqlitexplorer-1.1.1 → sqlitexplorer-1.1.2}/sqlitexplorer/cli.py +22 -11
  6. {sqlitexplorer-1.1.1 → sqlitexplorer-1.1.2/sqlitexplorer.egg-info}/PKG-INFO +11 -9
  7. {sqlitexplorer-1.1.1 → sqlitexplorer-1.1.2}/tests/test_charts.py +76 -1
  8. {sqlitexplorer-1.1.1 → sqlitexplorer-1.1.2}/LICENSE +0 -0
  9. {sqlitexplorer-1.1.1 → sqlitexplorer-1.1.2}/MANIFEST.in +0 -0
  10. {sqlitexplorer-1.1.1 → sqlitexplorer-1.1.2}/setup.cfg +0 -0
  11. {sqlitexplorer-1.1.1 → sqlitexplorer-1.1.2}/sqlitexplorer/__init__.py +0 -0
  12. {sqlitexplorer-1.1.1 → sqlitexplorer-1.1.2}/sqlitexplorer/__main__.py +0 -0
  13. {sqlitexplorer-1.1.1 → sqlitexplorer-1.1.2}/sqlitexplorer/completion.py +0 -0
  14. {sqlitexplorer-1.1.1 → sqlitexplorer-1.1.2}/sqlitexplorer/core.py +0 -0
  15. {sqlitexplorer-1.1.1 → sqlitexplorer-1.1.2}/sqlitexplorer/render.py +0 -0
  16. {sqlitexplorer-1.1.1 → sqlitexplorer-1.1.2}/sqlitexplorer/shell.py +0 -0
  17. {sqlitexplorer-1.1.1 → sqlitexplorer-1.1.2}/sqlitexplorer.egg-info/SOURCES.txt +0 -0
  18. {sqlitexplorer-1.1.1 → sqlitexplorer-1.1.2}/sqlitexplorer.egg-info/dependency_links.txt +0 -0
  19. {sqlitexplorer-1.1.1 → sqlitexplorer-1.1.2}/sqlitexplorer.egg-info/entry_points.txt +0 -0
  20. {sqlitexplorer-1.1.1 → sqlitexplorer-1.1.2}/sqlitexplorer.egg-info/requires.txt +0 -0
  21. {sqlitexplorer-1.1.1 → sqlitexplorer-1.1.2}/sqlitexplorer.egg-info/top_level.txt +0 -0
  22. {sqlitexplorer-1.1.1 → sqlitexplorer-1.1.2}/tests/conftest.py +0 -0
  23. {sqlitexplorer-1.1.1 → sqlitexplorer-1.1.2}/tests/test_cli.py +0 -0
  24. {sqlitexplorer-1.1.1 → sqlitexplorer-1.1.2}/tests/test_completion.py +0 -0
  25. {sqlitexplorer-1.1.1 → sqlitexplorer-1.1.2}/tests/test_core.py +0 -0
  26. {sqlitexplorer-1.1.1 → sqlitexplorer-1.1.2}/tests/test_render.py +0 -0
  27. {sqlitexplorer-1.1.1 → sqlitexplorer-1.1.2}/tests/test_shell.py +0 -0
@@ -1,6 +1,6 @@
1
1
  Metadata-Version: 2.4
2
2
  Name: sqlitexplorer
3
- Version: 1.1.1
3
+ Version: 1.1.2
4
4
  Summary: Explore SQLite databases from the terminal: tables, schema, stats, search, queries, charts, export/import and an interactive shell.
5
5
  Author-email: "Carlos A. Planchón" <carlosandresplanchonprestes@gmail.com>
6
6
  License-Expression: MIT
@@ -128,9 +128,9 @@ their values wrapped so that every column and label stays visible. Use
128
128
  Big results do not need to fit in memory: `--page` fetches only the requested
129
129
  page (in SQL for `show`, from the cursor for `query`), and the csv, tsv, json
130
130
  and markdown formats, `export` and `dump` are written row by row. The table
131
- format is one exception, since it needs every row to size its columns, and
132
- `chart` is the other: it reads the whole result before reducing it to what the
133
- canvas can show.
131
+ format is the exception, since it needs every row to size its columns.
132
+ `chart` reduces as it reads, so the memory it needs does not grow with the
133
+ table; `chart --no-resample` is the one that holds every row.
134
134
 
135
135
  ## Queries
136
136
 
@@ -163,11 +163,13 @@ stderr. `--kind` selects `line` (default), `scatter` or `hist` (first column
163
163
  only, `--bins`). `--height`, `--width`, `--x-label`, `--y-label` and `--color`
164
164
  adjust the drawing.
165
165
 
166
- Line and scatter charts are reduced to what the canvas can show before they
167
- reach plotille, keeping the minimum and the maximum of every column of braille
168
- dots, so spikes survive and the drawing no longer costs more as the query grows;
169
- the reduction is reported on stderr. `--no-resample` sends every row instead.
170
- Histograms are never reduced, since dropping rows would change the distribution.
166
+ Line and scatter charts are reduced to what the canvas can show, keeping the
167
+ minimum and the maximum of every column of braille dots, so spikes survive and
168
+ the true extremes keep the X of their own row. The rows are reduced as they are
169
+ read, in chunks, so a chart over millions of rows needs no more memory than one
170
+ over a thousand; the reduction is reported on stderr. `--no-resample` reads and
171
+ plots every row instead. Histograms are never reduced, since dropping rows would
172
+ change the distribution.
171
173
 
172
174
  ## Export, import, dump and diff
173
175
 
@@ -98,9 +98,9 @@ their values wrapped so that every column and label stays visible. Use
98
98
  Big results do not need to fit in memory: `--page` fetches only the requested
99
99
  page (in SQL for `show`, from the cursor for `query`), and the csv, tsv, json
100
100
  and markdown formats, `export` and `dump` are written row by row. The table
101
- format is one exception, since it needs every row to size its columns, and
102
- `chart` is the other: it reads the whole result before reducing it to what the
103
- canvas can show.
101
+ format is the exception, since it needs every row to size its columns.
102
+ `chart` reduces as it reads, so the memory it needs does not grow with the
103
+ table; `chart --no-resample` is the one that holds every row.
104
104
 
105
105
  ## Queries
106
106
 
@@ -133,11 +133,13 @@ stderr. `--kind` selects `line` (default), `scatter` or `hist` (first column
133
133
  only, `--bins`). `--height`, `--width`, `--x-label`, `--y-label` and `--color`
134
134
  adjust the drawing.
135
135
 
136
- Line and scatter charts are reduced to what the canvas can show before they
137
- reach plotille, keeping the minimum and the maximum of every column of braille
138
- dots, so spikes survive and the drawing no longer costs more as the query grows;
139
- the reduction is reported on stderr. `--no-resample` sends every row instead.
140
- Histograms are never reduced, since dropping rows would change the distribution.
136
+ Line and scatter charts are reduced to what the canvas can show, keeping the
137
+ minimum and the maximum of every column of braille dots, so spikes survive and
138
+ the true extremes keep the X of their own row. The rows are reduced as they are
139
+ read, in chunks, so a chart over millions of rows needs no more memory than one
140
+ over a thousand; the reduction is reported on stderr. `--no-resample` reads and
141
+ plots every row instead. Histograms are never reduced, since dropping rows would
142
+ change the distribution.
141
143
 
142
144
  ## Export, import, dump and diff
143
145
 
@@ -4,7 +4,7 @@ build-backend = "setuptools.build_meta"
4
4
 
5
5
  [project]
6
6
  name = "sqlitexplorer"
7
- version = "1.1.1"
7
+ version = "1.1.2"
8
8
  description = "Explore SQLite databases from the terminal: tables, schema, stats, search, queries, charts, export/import and an interactive shell."
9
9
  readme = "README.md"
10
10
  license = "MIT"
@@ -12,11 +12,12 @@ from contextlib import contextmanager
12
12
  from dataclasses import dataclass
13
13
  from datetime import datetime
14
14
  from enum import Enum
15
+ from itertools import islice
15
16
 
16
17
  import plotille
17
18
  import plotilleresample
18
19
 
19
- from sqlitexplorer.core import ExplorerError, ResultSet
20
+ from sqlitexplorer.core import ExplorerError, ResultSet, RowStream
20
21
 
21
22
  __all__ = [
22
23
  "ChartKind",
@@ -26,6 +27,7 @@ __all__ = [
26
27
  "render_histogram",
27
28
  "resample_series",
28
29
  "series_from_result",
30
+ "stream_series",
29
31
  ]
30
32
 
31
33
  PALETTE = ("red", "green", "yellow", "blue", "magenta", "cyan")
@@ -36,6 +38,10 @@ AXIS_WIDTH = 12
36
38
  # A tuple, not bool | int | float: the union would be rebuilt on every call,
37
39
  # and this runs once per value of the result.
38
40
  _NUMERIC = (bool, int, float)
41
+ # Rows read at a time when streaming, and how many reduced points may pile
42
+ # up before they are reduced again.
43
+ _CHUNK = 65536
44
+ _PILE = 8
39
45
 
40
46
 
41
47
  class ChartKind(str, Enum):
@@ -74,11 +80,13 @@ def _x_value(value: object) -> float | datetime | None:
74
80
  return None
75
81
 
76
82
 
77
- def series_from_result(result: ResultSet) -> tuple[list[Series], int]:
83
+ def series_from_result(result: ResultSet, *, require_rows: bool = True) -> tuple[list[Series], int]:
78
84
  """Split *result* into one series per numeric column after the first.
79
85
 
80
86
  Rows with a NULL in the X column or in any series are skipped; the second
81
- item of the returned tuple counts them.
87
+ item of the returned tuple counts them. With *require_rows* false an empty
88
+ result yields empty series instead of raising, which is what the streaming
89
+ reader needs for a chunk that holds nothing usable.
82
90
  """
83
91
  if len(result.columns) < 2:
84
92
  raise ExplorerError("need an X column and at least one numeric column")
@@ -106,7 +114,7 @@ def series_from_result(result: ResultSet) -> tuple[list[Series], int]:
106
114
  raise ExplorerError(f"column {name} is not numeric: {value!r}")
107
115
  append(number)
108
116
  keep_x(x)
109
- if not xs:
117
+ if require_rows and not xs:
110
118
  raise ExplorerError("no rows to plot")
111
119
  return [
112
120
  Series(label=name, x=xs, y=bucket) for name, bucket in zip(y_names, ys, strict=True)
@@ -134,6 +142,50 @@ def resample_series(
134
142
  return reduced
135
143
 
136
144
 
145
+ def stream_series(
146
+ stream: RowStream, *, kind: ChartKind, width: int, height: int
147
+ ) -> tuple[list[Series], int, int]:
148
+ """Reduce *stream* to the canvas without ever holding every row.
149
+
150
+ Rows are read in chunks, each chunk is reduced on its own and the reduced
151
+ points are reduced again as they pile up, so the memory a chart needs stops
152
+ growing with the size of the table. Returns the series, how many rows were
153
+ skipped for their NULLs and how many were read.
154
+ """
155
+ reduced: list[Series] = []
156
+ x_type: type | None = None
157
+ skipped = rows_read = 0
158
+ while True:
159
+ rows = list(islice(stream.rows, _CHUNK))
160
+ if not rows:
161
+ break
162
+ rows_read += len(rows)
163
+ chunk, chunk_skipped = series_from_result(
164
+ ResultSet(columns=stream.columns, rows=rows), require_rows=False
165
+ )
166
+ skipped += chunk_skipped
167
+ if not chunk[0].x:
168
+ continue
169
+ if x_type is None:
170
+ x_type = type(chunk[0].x[0])
171
+ elif not isinstance(chunk[0].x[0], x_type):
172
+ raise ExplorerError(f"column {stream.columns[0]} mixes numbers and dates")
173
+ chunk = resample_series(chunk, kind=kind, width=width, height=height)
174
+ reduced = (
175
+ [
176
+ Series(label=old.label, x=old.x + new.x, y=old.y + new.y)
177
+ for old, new in zip(reduced, chunk, strict=True)
178
+ ]
179
+ if reduced
180
+ else chunk
181
+ )
182
+ if len(reduced[0].x) > _PILE * len(chunk[0].x):
183
+ reduced = resample_series(reduced, kind=kind, width=width, height=height)
184
+ if not reduced or not reduced[0].x:
185
+ raise ExplorerError("no rows to plot")
186
+ return resample_series(reduced, kind=kind, width=width, height=height), skipped, rows_read
187
+
188
+
137
189
  def histogram_values(result: ResultSet) -> tuple[list[float], int]:
138
190
  """Numeric values of the first column of *result*, and how many NULLs were skipped."""
139
191
  if not result.columns:
@@ -28,8 +28,8 @@ from sqlitexplorer.charts import (
28
28
  histogram_values,
29
29
  render_chart,
30
30
  render_histogram,
31
- resample_series,
32
31
  series_from_result,
32
+ stream_series,
33
33
  )
34
34
  from sqlitexplorer.completion import complete_table
35
35
  from sqlitexplorer.core import (
@@ -666,12 +666,13 @@ def chart(
666
666
  parameters = _parameters(params)
667
667
  with _reporting_errors(), open_database(database) as db:
668
668
  text = sys.stdin.read() if sql == "-" else sql
669
- result = db.execute(text, parameters if parameters else ())
670
- if not result.returns_rows:
671
- _fail("the statement returned no rows")
669
+ bound = parameters if parameters else ()
672
670
  use_color = resolve_color(color)
673
671
  screen = width if width is not None else shutil.get_terminal_size().columns
674
672
  if kind is ChartKind.HIST:
673
+ result = db.execute(text, bound)
674
+ if not result.returns_rows:
675
+ _fail("the statement returned no rows")
675
676
  values, skipped = histogram_values(result)
676
677
  drawing = render_histogram(
677
678
  values,
@@ -683,13 +684,23 @@ def chart(
683
684
  y_label=y_label or "count",
684
685
  )
685
686
  else:
686
- series, skipped = series_from_result(result)
687
687
  if resample:
688
- rows = len(series[0].x)
689
- series = resample_series(series, kind=kind, width=screen, height=height)
690
- points = len(series[0].x)
691
- if points < rows:
692
- typer.echo(f"resampled {rows} rows to {points} points", err=True)
688
+ # Reduces as it reads, so the rows never pile up in memory.
689
+ stream = db.stream(text, bound)
690
+ if not stream.returns_rows:
691
+ _fail("the statement returned no rows")
692
+ columns = stream.columns
693
+ series, skipped, read = stream_series(
694
+ stream, kind=kind, width=screen, height=height
695
+ )
696
+ if len(series[0].x) < read:
697
+ typer.echo(f"resampled {read} rows to {len(series[0].x)} points", err=True)
698
+ else:
699
+ result = db.execute(text, bound)
700
+ if not result.returns_rows:
701
+ _fail("the statement returned no rows")
702
+ columns = result.columns
703
+ series, skipped = series_from_result(result)
693
704
  default_y = series[0].label if len(series) == 1 else "value"
694
705
  drawing = render_chart(
695
706
  series,
@@ -697,7 +708,7 @@ def chart(
697
708
  width=screen,
698
709
  height=height,
699
710
  color=use_color,
700
- x_label=x_label or result.columns[0],
711
+ x_label=x_label or columns[0],
701
712
  y_label=y_label or default_y,
702
713
  )
703
714
  if skipped:
@@ -1,6 +1,6 @@
1
1
  Metadata-Version: 2.4
2
2
  Name: sqlitexplorer
3
- Version: 1.1.1
3
+ Version: 1.1.2
4
4
  Summary: Explore SQLite databases from the terminal: tables, schema, stats, search, queries, charts, export/import and an interactive shell.
5
5
  Author-email: "Carlos A. Planchón" <carlosandresplanchonprestes@gmail.com>
6
6
  License-Expression: MIT
@@ -128,9 +128,9 @@ their values wrapped so that every column and label stays visible. Use
128
128
  Big results do not need to fit in memory: `--page` fetches only the requested
129
129
  page (in SQL for `show`, from the cursor for `query`), and the csv, tsv, json
130
130
  and markdown formats, `export` and `dump` are written row by row. The table
131
- format is one exception, since it needs every row to size its columns, and
132
- `chart` is the other: it reads the whole result before reducing it to what the
133
- canvas can show.
131
+ format is the exception, since it needs every row to size its columns.
132
+ `chart` reduces as it reads, so the memory it needs does not grow with the
133
+ table; `chart --no-resample` is the one that holds every row.
134
134
 
135
135
  ## Queries
136
136
 
@@ -163,11 +163,13 @@ stderr. `--kind` selects `line` (default), `scatter` or `hist` (first column
163
163
  only, `--bins`). `--height`, `--width`, `--x-label`, `--y-label` and `--color`
164
164
  adjust the drawing.
165
165
 
166
- Line and scatter charts are reduced to what the canvas can show before they
167
- reach plotille, keeping the minimum and the maximum of every column of braille
168
- dots, so spikes survive and the drawing no longer costs more as the query grows;
169
- the reduction is reported on stderr. `--no-resample` sends every row instead.
170
- Histograms are never reduced, since dropping rows would change the distribution.
166
+ Line and scatter charts are reduced to what the canvas can show, keeping the
167
+ minimum and the maximum of every column of braille dots, so spikes survive and
168
+ the true extremes keep the X of their own row. The rows are reduced as they are
169
+ read, in chunks, so a chart over millions of rows needs no more memory than one
170
+ over a thousand; the reduction is reported on stderr. `--no-resample` reads and
171
+ plots every row instead. Histograms are never reduced, since dropping rows would
172
+ change the distribution.
171
173
 
172
174
  ## Export, import, dump and diff
173
175
 
@@ -13,8 +13,9 @@ from sqlitexplorer.charts import (
13
13
  render_histogram,
14
14
  resample_series,
15
15
  series_from_result,
16
+ stream_series,
16
17
  )
17
- from sqlitexplorer.core import ExplorerError, ResultSet
18
+ from sqlitexplorer.core import ExplorerError, ResultSet, RowStream
18
19
 
19
20
 
20
21
  def braille(text: str) -> bool:
@@ -152,3 +153,77 @@ def test_resample_series_scatter_keeps_more_points_than_a_line() -> None:
152
153
  line = resample_series(series, kind=ChartKind.LINE, width=80, height=15)
153
154
  scatter = resample_series(series, kind=ChartKind.SCATTER, width=80, height=15)
154
155
  assert len(scatter[0].x) > len(line[0].x)
156
+
157
+
158
+ def _stream(columns: tuple[str, ...], rows: list[tuple]) -> RowStream:
159
+ return RowStream(columns=columns, rows=iter(rows))
160
+
161
+
162
+ def test_stream_series_keeps_the_envelope_and_the_real_pairs(
163
+ monkeypatch: pytest.MonkeyPatch,
164
+ ) -> None:
165
+ monkeypatch.setattr("sqlitexplorer.charts._CHUNK", 50)
166
+ rows: list[tuple] = [(i, 0.0) for i in range(5000)]
167
+ rows[1234] = (1234, 999.0)
168
+ rows[4321] = (4321, -999.0)
169
+ series, skipped, read = stream_series(
170
+ _stream(("x", "y"), rows), kind=ChartKind.LINE, width=80, height=15
171
+ )
172
+ assert read == 5000
173
+ assert skipped == 0
174
+ assert len(series[0].x) < 5000
175
+ assert max(series[0].y) == 999.0
176
+ assert min(series[0].y) == -999.0
177
+ # The extremes keep the X of their own row, not a neighbour's.
178
+ assert series[0].x[series[0].y.index(999.0)] == 1234
179
+ assert series[0].x[series[0].y.index(-999.0)] == 4321
180
+ assert series[0].x == sorted(series[0].x)
181
+
182
+
183
+ def test_stream_series_matches_the_collected_path_on_one_chunk() -> None:
184
+ rows = [(i, float(i % 97)) for i in range(4000)]
185
+ collected, _ = series_from_result(ResultSet(columns=("x", "y"), rows=rows))
186
+ collected = resample_series(collected, kind=ChartKind.LINE, width=80, height=15)
187
+ streamed, _, read = stream_series(
188
+ _stream(("x", "y"), rows), kind=ChartKind.LINE, width=80, height=15
189
+ )
190
+ assert read == 4000
191
+ assert streamed[0].x == collected[0].x
192
+ assert streamed[0].y == collected[0].y
193
+
194
+
195
+ def test_stream_series_skips_nulls_across_chunks(monkeypatch: pytest.MonkeyPatch) -> None:
196
+ monkeypatch.setattr("sqlitexplorer.charts._CHUNK", 10)
197
+ rows: list[tuple] = [(i, None) if 10 <= i < 20 else (i, float(i)) for i in range(100)]
198
+ series, skipped, read = stream_series(
199
+ _stream(("x", "y"), rows), kind=ChartKind.LINE, width=80, height=15
200
+ )
201
+ assert (read, skipped) == (100, 10)
202
+ assert min(series[0].y) == 0.0
203
+ assert max(series[0].y) == 99.0
204
+
205
+
206
+ def test_stream_series_handles_dates_and_rejects_a_mix(
207
+ monkeypatch: pytest.MonkeyPatch,
208
+ ) -> None:
209
+ monkeypatch.setattr("sqlitexplorer.charts._CHUNK", 10)
210
+ dates = [(f"2020-01-01T00:00:{i:02d}", float(i)) for i in range(60)]
211
+ series, _, read = stream_series(
212
+ _stream(("t", "v"), dates), kind=ChartKind.LINE, width=80, height=15
213
+ )
214
+ assert read == 60
215
+ assert all(isinstance(value, datetime) for value in series[0].x)
216
+ mixed = dates + [(5, 1.0)] * 10
217
+ with pytest.raises(ExplorerError, match="mixes numbers and dates"):
218
+ stream_series(_stream(("t", "v"), mixed), kind=ChartKind.LINE, width=80, height=15)
219
+
220
+
221
+ def test_stream_series_rejects_an_empty_stream() -> None:
222
+ with pytest.raises(ExplorerError, match="no rows to plot"):
223
+ stream_series(_stream(("x", "y"), []), kind=ChartKind.LINE, width=80, height=15)
224
+
225
+
226
+ def test_series_from_result_can_allow_an_empty_result() -> None:
227
+ series, skipped = series_from_result(ResultSet(columns=("x", "y")), require_rows=False)
228
+ assert [item.label for item in series] == ["y"]
229
+ assert series[0].x == [] and skipped == 0
File without changes
File without changes
File without changes