sqlitexplorer 1.0.0__py3-none-any.whl
This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
- sqlitexplorer/__init__.py +10 -0
- sqlitexplorer/__main__.py +6 -0
- sqlitexplorer/charts.py +203 -0
- sqlitexplorer/cli.py +857 -0
- sqlitexplorer/completion.py +22 -0
- sqlitexplorer/core.py +700 -0
- sqlitexplorer/render.py +508 -0
- sqlitexplorer/shell.py +254 -0
- sqlitexplorer-1.0.0.dist-info/METADATA +229 -0
- sqlitexplorer-1.0.0.dist-info/RECORD +14 -0
- sqlitexplorer-1.0.0.dist-info/WHEEL +5 -0
- sqlitexplorer-1.0.0.dist-info/entry_points.txt +2 -0
- sqlitexplorer-1.0.0.dist-info/licenses/LICENSE +21 -0
- sqlitexplorer-1.0.0.dist-info/top_level.txt +1 -0
sqlitexplorer/render.py
ADDED
|
@@ -0,0 +1,508 @@
|
|
|
1
|
+
"""Presentation layer: output formats, pagination, pager, colors and the shared emit path.
|
|
2
|
+
|
|
3
|
+
Every result set printed by the CLI goes through :func:`emit`, so that
|
|
4
|
+
``--format``, ``--null``, ``--truncate``, ``--page``, ``--pager``, ``--color``
|
|
5
|
+
and ``--width`` behave identically in every command. This is also the only
|
|
6
|
+
module that talks to outfancy.
|
|
7
|
+
"""
|
|
8
|
+
|
|
9
|
+
from __future__ import annotations
|
|
10
|
+
|
|
11
|
+
import base64
|
|
12
|
+
import csv
|
|
13
|
+
import io
|
|
14
|
+
import json
|
|
15
|
+
import os
|
|
16
|
+
import re
|
|
17
|
+
import shlex
|
|
18
|
+
import shutil
|
|
19
|
+
import subprocess
|
|
20
|
+
import sys
|
|
21
|
+
import textwrap
|
|
22
|
+
from collections.abc import Sequence
|
|
23
|
+
from dataclasses import dataclass
|
|
24
|
+
from enum import Enum
|
|
25
|
+
from typing import TextIO
|
|
26
|
+
|
|
27
|
+
import outfancy.table
|
|
28
|
+
import typer
|
|
29
|
+
|
|
30
|
+
from sqlitexplorer.core import ExplorerError, ResultSet, RowStream
|
|
31
|
+
|
|
32
|
+
__all__ = [
|
|
33
|
+
"OutputFormat",
|
|
34
|
+
"OutputOptions",
|
|
35
|
+
"coerce_rows",
|
|
36
|
+
"emit",
|
|
37
|
+
"emit_stream",
|
|
38
|
+
"fit_widths",
|
|
39
|
+
"format_value",
|
|
40
|
+
"infer_types",
|
|
41
|
+
"ok_message",
|
|
42
|
+
"paginate",
|
|
43
|
+
"parse_number",
|
|
44
|
+
"parse_rows",
|
|
45
|
+
"render",
|
|
46
|
+
"resolve_color",
|
|
47
|
+
"stdout_is_tty",
|
|
48
|
+
"strip_ansi",
|
|
49
|
+
"write_rows",
|
|
50
|
+
]
|
|
51
|
+
|
|
52
|
+
ELLIPSIS = "…"
|
|
53
|
+
_ANSI = re.compile(r"\x1b\[[0-9;]*m")
|
|
54
|
+
_INTEGER = re.compile(r"^[+-]?\d+$")
|
|
55
|
+
_REAL = re.compile(r"^[+-]?(\d+\.\d*|\.\d+|\d+)([eE][+-]?\d+)?$")
|
|
56
|
+
|
|
57
|
+
|
|
58
|
+
class OutputFormat(str, Enum):
|
|
59
|
+
TABLE = "table"
|
|
60
|
+
CSV = "csv"
|
|
61
|
+
TSV = "tsv"
|
|
62
|
+
JSON = "json"
|
|
63
|
+
MARKDOWN = "markdown"
|
|
64
|
+
|
|
65
|
+
|
|
66
|
+
@dataclass
|
|
67
|
+
class OutputOptions:
|
|
68
|
+
"""How a result set should be printed. Mutable: the shell changes it at runtime."""
|
|
69
|
+
|
|
70
|
+
format: OutputFormat = OutputFormat.TABLE
|
|
71
|
+
null: str = "NULL"
|
|
72
|
+
truncate: int | None = None
|
|
73
|
+
page: int | None = None
|
|
74
|
+
page_size: int | None = None
|
|
75
|
+
pager: bool = False
|
|
76
|
+
color: bool | None = None
|
|
77
|
+
width: int | None = None
|
|
78
|
+
|
|
79
|
+
|
|
80
|
+
# --- Values -------------------------------------------------------------------
|
|
81
|
+
|
|
82
|
+
|
|
83
|
+
def format_value(value: object, *, null: str = "NULL", truncate: int | None = None) -> object:
|
|
84
|
+
"""Show NULL and BLOB values the way SQLite writes them, optionally truncated."""
|
|
85
|
+
if value is None:
|
|
86
|
+
return null
|
|
87
|
+
if isinstance(value, bytes):
|
|
88
|
+
value = f"X'{value.hex().upper()}'"
|
|
89
|
+
if truncate is not None and isinstance(value, str) and len(value) > truncate:
|
|
90
|
+
return value[: max(truncate - 1, 0)] + ELLIPSIS
|
|
91
|
+
return value
|
|
92
|
+
|
|
93
|
+
|
|
94
|
+
def _formatted_rows(result: ResultSet, *, null: str, truncate: int | None) -> list[tuple]:
|
|
95
|
+
return [
|
|
96
|
+
tuple(format_value(v, null=null, truncate=truncate) for v in row) for row in result.rows
|
|
97
|
+
]
|
|
98
|
+
|
|
99
|
+
|
|
100
|
+
def strip_ansi(text: str) -> str:
|
|
101
|
+
return _ANSI.sub("", text)
|
|
102
|
+
|
|
103
|
+
|
|
104
|
+
def parse_number(text: str) -> int | float | str:
|
|
105
|
+
"""Turn *text* into an int or a float when it looks like one, else keep it."""
|
|
106
|
+
stripped = text.strip()
|
|
107
|
+
if _INTEGER.match(stripped):
|
|
108
|
+
return int(stripped)
|
|
109
|
+
if _REAL.match(stripped):
|
|
110
|
+
return float(stripped)
|
|
111
|
+
return text
|
|
112
|
+
|
|
113
|
+
|
|
114
|
+
def ok_message(result: ResultSet | RowStream) -> str:
|
|
115
|
+
"""What to print for a statement that returned no rows."""
|
|
116
|
+
if result.rowcount >= 0:
|
|
117
|
+
plural = "" if result.rowcount == 1 else "s"
|
|
118
|
+
return f"OK ({result.rowcount} row{plural} affected)"
|
|
119
|
+
return "OK"
|
|
120
|
+
|
|
121
|
+
|
|
122
|
+
# --- Renderers ----------------------------------------------------------------
|
|
123
|
+
|
|
124
|
+
|
|
125
|
+
def _natural_widths(columns: Sequence[str], rows: Sequence[tuple]) -> list[int]:
|
|
126
|
+
"""Width each column needs to show its label and every value in full."""
|
|
127
|
+
widths = [len(label) for label in columns]
|
|
128
|
+
for row in rows:
|
|
129
|
+
for index, value in enumerate(row):
|
|
130
|
+
longest = max((len(line) for line in str(value).splitlines()), default=0)
|
|
131
|
+
widths[index] = max(widths[index], longest)
|
|
132
|
+
return widths
|
|
133
|
+
|
|
134
|
+
|
|
135
|
+
def fit_widths(columns: Sequence[str], natural: Sequence[int], available: int) -> list[int] | None:
|
|
136
|
+
"""Column widths that keep every label intact within *available* characters.
|
|
137
|
+
|
|
138
|
+
Columns get their natural width when everything fits. Otherwise the widest
|
|
139
|
+
columns are narrowed first, down to a common cap, so that every column and
|
|
140
|
+
label stays visible and long values wrap. Returns None when not even the
|
|
141
|
+
labels fit, in which case the caller lets outfancy decide.
|
|
142
|
+
"""
|
|
143
|
+
minimum = [max(len(label), 1) for label in columns]
|
|
144
|
+
if not columns or sum(minimum) > available:
|
|
145
|
+
return None
|
|
146
|
+
wanted = [max(n, m) for n, m in zip(natural, minimum, strict=True)]
|
|
147
|
+
if sum(wanted) <= available:
|
|
148
|
+
return wanted
|
|
149
|
+
|
|
150
|
+
def total(cap: int) -> int:
|
|
151
|
+
return sum(max(m, min(n, cap)) for n, m in zip(wanted, minimum, strict=True))
|
|
152
|
+
|
|
153
|
+
# Largest cap whose total still fits.
|
|
154
|
+
low, high = 0, max(wanted)
|
|
155
|
+
while low < high:
|
|
156
|
+
middle = (low + high + 1) // 2
|
|
157
|
+
if total(middle) <= available:
|
|
158
|
+
low = middle
|
|
159
|
+
else:
|
|
160
|
+
high = middle - 1
|
|
161
|
+
widths = [max(m, min(n, low)) for n, m in zip(wanted, minimum, strict=True)]
|
|
162
|
+
slack = available - sum(widths)
|
|
163
|
+
for index in sorted(range(len(widths)), key=lambda i: wanted[i], reverse=True):
|
|
164
|
+
if slack <= 0:
|
|
165
|
+
break
|
|
166
|
+
if widths[index] < wanted[index]:
|
|
167
|
+
widths[index] += 1
|
|
168
|
+
slack -= 1
|
|
169
|
+
return widths
|
|
170
|
+
|
|
171
|
+
|
|
172
|
+
def render_table(
|
|
173
|
+
result: ResultSet,
|
|
174
|
+
*,
|
|
175
|
+
width: int | None = None,
|
|
176
|
+
empty: str = "(no rows)",
|
|
177
|
+
null: str = "NULL",
|
|
178
|
+
truncate: int | None = None,
|
|
179
|
+
) -> str:
|
|
180
|
+
"""Render *result* as an outfancy table.
|
|
181
|
+
|
|
182
|
+
outfancy sizes columns from the data alone, clips labels that do not fit
|
|
183
|
+
and drops whole columns when the table is too wide. The widths are
|
|
184
|
+
therefore computed here (see :func:`fit_widths`) and passed explicitly;
|
|
185
|
+
outfancy only takes over when not even the labels fit on the screen.
|
|
186
|
+
"""
|
|
187
|
+
table = outfancy.table.Table()
|
|
188
|
+
table.set_empty_string(empty)
|
|
189
|
+
rows = _formatted_rows(result, null=null, truncate=truncate)
|
|
190
|
+
columns = list(result.columns)
|
|
191
|
+
|
|
192
|
+
screen = width if width is not None else shutil.get_terminal_size().columns
|
|
193
|
+
# outfancy keeps a two-column margin and one separator per column.
|
|
194
|
+
available = screen - 2 - len(columns)
|
|
195
|
+
widths = fit_widths(columns, _natural_widths(columns, rows), available)
|
|
196
|
+
return table.render(data=rows, label_list=columns, width=widths, screen_x=width)
|
|
197
|
+
|
|
198
|
+
|
|
199
|
+
def render_csv(
|
|
200
|
+
result: ResultSet, *, delimiter: str = ",", null: str = "NULL", truncate: int | None = None
|
|
201
|
+
) -> str:
|
|
202
|
+
buffer = io.StringIO()
|
|
203
|
+
writer = csv.writer(buffer, delimiter=delimiter, lineterminator="\n")
|
|
204
|
+
writer.writerow(result.columns)
|
|
205
|
+
writer.writerows(_formatted_rows(result, null=null, truncate=truncate))
|
|
206
|
+
return buffer.getvalue().rstrip("\n")
|
|
207
|
+
|
|
208
|
+
|
|
209
|
+
def _json_value(value: object) -> object:
|
|
210
|
+
if isinstance(value, bytes):
|
|
211
|
+
return base64.b64encode(value).decode("ascii")
|
|
212
|
+
return value
|
|
213
|
+
|
|
214
|
+
|
|
215
|
+
def render_json(result: ResultSet) -> str:
|
|
216
|
+
records = [dict(zip(result.columns, map(_json_value, row), strict=True)) for row in result.rows]
|
|
217
|
+
return json.dumps(records, ensure_ascii=False, indent=2)
|
|
218
|
+
|
|
219
|
+
|
|
220
|
+
def _markdown_cell(value: object) -> str:
|
|
221
|
+
text = str(value).replace("|", "\\|")
|
|
222
|
+
return " ".join(text.splitlines()) if "\n" in text or "\r" in text else text
|
|
223
|
+
|
|
224
|
+
|
|
225
|
+
def _markdown_row(values: Sequence[object]) -> str:
|
|
226
|
+
return "| " + " | ".join(_markdown_cell(value) for value in values) + " |"
|
|
227
|
+
|
|
228
|
+
|
|
229
|
+
def _markdown_header(columns: Sequence[str]) -> list[str]:
|
|
230
|
+
return [_markdown_row(columns), "|" + "|".join(" --- " for _ in columns) + "|"]
|
|
231
|
+
|
|
232
|
+
|
|
233
|
+
def render_markdown(result: ResultSet, *, null: str = "NULL", truncate: int | None = None) -> str:
|
|
234
|
+
lines = _markdown_header(result.columns)
|
|
235
|
+
lines.extend(
|
|
236
|
+
_markdown_row(row) for row in _formatted_rows(result, null=null, truncate=truncate)
|
|
237
|
+
)
|
|
238
|
+
return "\n".join(lines)
|
|
239
|
+
|
|
240
|
+
|
|
241
|
+
def render(result: ResultSet, options: OutputOptions, *, empty: str = "(no rows)") -> str:
|
|
242
|
+
"""Render *result* in the format selected by *options*."""
|
|
243
|
+
fmt = OutputFormat(options.format)
|
|
244
|
+
if fmt is OutputFormat.TABLE:
|
|
245
|
+
return render_table(
|
|
246
|
+
result, width=options.width, empty=empty, null=options.null, truncate=options.truncate
|
|
247
|
+
)
|
|
248
|
+
if fmt is OutputFormat.CSV:
|
|
249
|
+
return render_csv(result, null=options.null, truncate=options.truncate)
|
|
250
|
+
if fmt is OutputFormat.TSV:
|
|
251
|
+
return render_csv(result, delimiter="\t", null=options.null, truncate=options.truncate)
|
|
252
|
+
if fmt is OutputFormat.JSON:
|
|
253
|
+
return render_json(result)
|
|
254
|
+
return render_markdown(result, null=options.null, truncate=options.truncate)
|
|
255
|
+
|
|
256
|
+
|
|
257
|
+
# --- Pagination, colors and pager --------------------------------------------
|
|
258
|
+
|
|
259
|
+
|
|
260
|
+
def default_page_size() -> int:
|
|
261
|
+
"""Rows that fit on the terminal, leaving room for the labels and the footer."""
|
|
262
|
+
return max(1, shutil.get_terminal_size().lines - 4)
|
|
263
|
+
|
|
264
|
+
|
|
265
|
+
def paginate(
|
|
266
|
+
result: ResultSet, *, page: int | None = None, page_size: int | None = None
|
|
267
|
+
) -> tuple[ResultSet, str | None]:
|
|
268
|
+
"""Slice *result* to one page. Returns the page and a footer, or the result untouched."""
|
|
269
|
+
if page is None and page_size is None:
|
|
270
|
+
return result, None
|
|
271
|
+
size = page_size or default_page_size()
|
|
272
|
+
number = page or 1
|
|
273
|
+
if result.total is not None:
|
|
274
|
+
# core already fetched just this page; only the footer is missing.
|
|
275
|
+
pages = max(1, -(-result.total // size))
|
|
276
|
+
if number > pages:
|
|
277
|
+
raise ExplorerError(f"page {number} is out of range (1-{pages})")
|
|
278
|
+
return result, f"page {number} of {pages} ({result.total} rows)"
|
|
279
|
+
pages = max(1, -(-len(result.rows) // size))
|
|
280
|
+
if number > pages:
|
|
281
|
+
raise ExplorerError(f"page {number} is out of range (1-{pages})")
|
|
282
|
+
start = (number - 1) * size
|
|
283
|
+
sliced = ResultSet(
|
|
284
|
+
columns=result.columns, rows=result.rows[start : start + size], rowcount=result.rowcount
|
|
285
|
+
)
|
|
286
|
+
return sliced, f"page {number} of {pages} ({len(result.rows)} rows)"
|
|
287
|
+
|
|
288
|
+
|
|
289
|
+
def stdout_is_tty() -> bool:
|
|
290
|
+
try:
|
|
291
|
+
return sys.stdout.isatty()
|
|
292
|
+
except (AttributeError, ValueError):
|
|
293
|
+
return False
|
|
294
|
+
|
|
295
|
+
|
|
296
|
+
def resolve_color(color: bool | None) -> bool:
|
|
297
|
+
"""An explicit flag wins; otherwise color only on a terminal and unless NO_COLOR is set."""
|
|
298
|
+
if color is not None:
|
|
299
|
+
return color
|
|
300
|
+
return not os.environ.get("NO_COLOR") and stdout_is_tty()
|
|
301
|
+
|
|
302
|
+
|
|
303
|
+
def page_through(text: str) -> bool:
|
|
304
|
+
"""Send *text* to ``$PAGER`` (default ``less -R``). Returns False when not paged."""
|
|
305
|
+
if not stdout_is_tty():
|
|
306
|
+
return False
|
|
307
|
+
command = shlex.split(os.environ.get("PAGER") or "less -R")
|
|
308
|
+
if not command:
|
|
309
|
+
return False
|
|
310
|
+
try:
|
|
311
|
+
subprocess.run(command, input=text, text=True, check=False)
|
|
312
|
+
except FileNotFoundError:
|
|
313
|
+
return False
|
|
314
|
+
return True
|
|
315
|
+
|
|
316
|
+
|
|
317
|
+
def emit(result: ResultSet, options: OutputOptions, *, empty: str = "(no rows)") -> None:
|
|
318
|
+
"""Print *result* honouring every output option. The single print path of the CLI."""
|
|
319
|
+
page, footer = paginate(result, page=options.page, page_size=options.page_size)
|
|
320
|
+
text = render(page, options, empty=empty)
|
|
321
|
+
use_color = resolve_color(options.color)
|
|
322
|
+
if not use_color:
|
|
323
|
+
text = strip_ansi(text)
|
|
324
|
+
if not (options.pager and page_through(text)):
|
|
325
|
+
typer.echo(text, color=use_color)
|
|
326
|
+
if footer:
|
|
327
|
+
typer.echo(footer, err=True)
|
|
328
|
+
|
|
329
|
+
|
|
330
|
+
# --- Streaming ------------------------------------------------------------------
|
|
331
|
+
|
|
332
|
+
|
|
333
|
+
def _write_csv(stream: RowStream, options: OutputOptions, out: TextIO, delimiter: str) -> int:
|
|
334
|
+
writer = csv.writer(out, delimiter=delimiter, lineterminator="\n")
|
|
335
|
+
writer.writerow(stream.columns)
|
|
336
|
+
count = 0
|
|
337
|
+
for row in stream.rows:
|
|
338
|
+
writer.writerow(
|
|
339
|
+
tuple(format_value(v, null=options.null, truncate=options.truncate) for v in row)
|
|
340
|
+
)
|
|
341
|
+
count += 1
|
|
342
|
+
return count
|
|
343
|
+
|
|
344
|
+
|
|
345
|
+
def _write_json(stream: RowStream, out: TextIO) -> int:
|
|
346
|
+
count = 0
|
|
347
|
+
for row in stream.rows:
|
|
348
|
+
record = dict(zip(stream.columns, map(_json_value, row), strict=True))
|
|
349
|
+
out.write("[\n" if count == 0 else ",\n")
|
|
350
|
+
out.write(textwrap.indent(json.dumps(record, ensure_ascii=False, indent=2), " "))
|
|
351
|
+
count += 1
|
|
352
|
+
out.write("[]\n" if count == 0 else "\n]\n")
|
|
353
|
+
return count
|
|
354
|
+
|
|
355
|
+
|
|
356
|
+
def _write_markdown(stream: RowStream, options: OutputOptions, out: TextIO) -> int:
|
|
357
|
+
out.write("\n".join(_markdown_header(stream.columns)) + "\n")
|
|
358
|
+
count = 0
|
|
359
|
+
for row in stream.rows:
|
|
360
|
+
values = tuple(format_value(v, null=options.null, truncate=options.truncate) for v in row)
|
|
361
|
+
out.write(_markdown_row(values) + "\n")
|
|
362
|
+
count += 1
|
|
363
|
+
return count
|
|
364
|
+
|
|
365
|
+
|
|
366
|
+
def write_rows(
|
|
367
|
+
stream: RowStream, options: OutputOptions, out: TextIO, *, empty: str = "(no rows)"
|
|
368
|
+
) -> int:
|
|
369
|
+
"""Write *stream* to *out* row by row. Returns how many rows were written.
|
|
370
|
+
|
|
371
|
+
The output is identical to :func:`render` followed by a newline. The table
|
|
372
|
+
format needs every row to size its columns, so it is the one format that
|
|
373
|
+
loads the whole stream first.
|
|
374
|
+
"""
|
|
375
|
+
fmt = OutputFormat(options.format)
|
|
376
|
+
if fmt is OutputFormat.TABLE:
|
|
377
|
+
result = stream.collect()
|
|
378
|
+
text = render_table(
|
|
379
|
+
result, width=options.width, empty=empty, null=options.null, truncate=options.truncate
|
|
380
|
+
)
|
|
381
|
+
out.write(strip_ansi(text) + "\n")
|
|
382
|
+
count = len(result.rows)
|
|
383
|
+
elif fmt is OutputFormat.JSON:
|
|
384
|
+
count = _write_json(stream, out)
|
|
385
|
+
elif fmt is OutputFormat.MARKDOWN:
|
|
386
|
+
count = _write_markdown(stream, options, out)
|
|
387
|
+
else:
|
|
388
|
+
count = _write_csv(stream, options, out, "\t" if fmt is OutputFormat.TSV else ",")
|
|
389
|
+
out.flush()
|
|
390
|
+
return count
|
|
391
|
+
|
|
392
|
+
|
|
393
|
+
def emit_stream(stream: RowStream, options: OutputOptions, *, empty: str = "(no rows)") -> int:
|
|
394
|
+
"""Print *stream*, row by row when the format allows it. Returns the rows printed.
|
|
395
|
+
|
|
396
|
+
The table format, the pager and pagination need the whole result, so
|
|
397
|
+
those cases fall back to :func:`emit`.
|
|
398
|
+
"""
|
|
399
|
+
fmt = OutputFormat(options.format)
|
|
400
|
+
needs_everything = (
|
|
401
|
+
fmt is OutputFormat.TABLE
|
|
402
|
+
or options.pager
|
|
403
|
+
or options.page is not None
|
|
404
|
+
or options.page_size is not None
|
|
405
|
+
)
|
|
406
|
+
if needs_everything:
|
|
407
|
+
result = stream.collect()
|
|
408
|
+
emit(result, options, empty=empty)
|
|
409
|
+
return len(result.rows)
|
|
410
|
+
return write_rows(stream, options, sys.stdout, empty=empty)
|
|
411
|
+
|
|
412
|
+
|
|
413
|
+
# --- Parsing (import) ---------------------------------------------------------
|
|
414
|
+
|
|
415
|
+
|
|
416
|
+
def parse_rows(
|
|
417
|
+
text: str, fmt: OutputFormat, *, delimiter: str | None = None
|
|
418
|
+
) -> tuple[list[str], list[list[object]]]:
|
|
419
|
+
"""Read ``(headers, rows)`` from csv/tsv text or from a JSON list of objects."""
|
|
420
|
+
if fmt is OutputFormat.JSON:
|
|
421
|
+
return _parse_json_rows(text)
|
|
422
|
+
if fmt not in (OutputFormat.CSV, OutputFormat.TSV):
|
|
423
|
+
raise ExplorerError(f"cannot import from the {fmt.value} format")
|
|
424
|
+
separator = delimiter or ("\t" if fmt is OutputFormat.TSV else ",")
|
|
425
|
+
reader = csv.reader(io.StringIO(text), delimiter=separator)
|
|
426
|
+
try:
|
|
427
|
+
headers = next(reader)
|
|
428
|
+
except StopIteration:
|
|
429
|
+
raise ExplorerError("empty file") from None
|
|
430
|
+
if not any(header.strip() for header in headers):
|
|
431
|
+
raise ExplorerError("empty file")
|
|
432
|
+
rows: list[list[object]] = []
|
|
433
|
+
for number, record in enumerate(reader, start=2):
|
|
434
|
+
if not record:
|
|
435
|
+
continue
|
|
436
|
+
if len(record) > len(headers):
|
|
437
|
+
raise ExplorerError(f"row {number} has {len(record)} values, expected {len(headers)}")
|
|
438
|
+
record = record + [""] * (len(headers) - len(record))
|
|
439
|
+
rows.append([None if cell == "" else cell for cell in record])
|
|
440
|
+
return headers, rows
|
|
441
|
+
|
|
442
|
+
|
|
443
|
+
def _parse_json_rows(text: str) -> tuple[list[str], list[list[object]]]:
|
|
444
|
+
try:
|
|
445
|
+
data = json.loads(text)
|
|
446
|
+
except json.JSONDecodeError as error:
|
|
447
|
+
raise ExplorerError(f"invalid JSON: {error}") from error
|
|
448
|
+
if not isinstance(data, list) or not all(isinstance(item, dict) for item in data):
|
|
449
|
+
raise ExplorerError("JSON input must be a list of objects")
|
|
450
|
+
headers: list[str] = []
|
|
451
|
+
for item in data:
|
|
452
|
+
headers.extend(key for key in item if key not in headers)
|
|
453
|
+
if not headers:
|
|
454
|
+
raise ExplorerError("empty file")
|
|
455
|
+
rows = [[_plain_json_value(item.get(key)) for key in headers] for item in data]
|
|
456
|
+
return headers, rows
|
|
457
|
+
|
|
458
|
+
|
|
459
|
+
def _plain_json_value(value: object) -> object:
|
|
460
|
+
if isinstance(value, bool):
|
|
461
|
+
return int(value)
|
|
462
|
+
if isinstance(value, (dict, list)):
|
|
463
|
+
return json.dumps(value, ensure_ascii=False)
|
|
464
|
+
return value
|
|
465
|
+
|
|
466
|
+
|
|
467
|
+
def _classify(value: object) -> str:
|
|
468
|
+
if isinstance(value, bool | int):
|
|
469
|
+
return "INTEGER"
|
|
470
|
+
if isinstance(value, float):
|
|
471
|
+
return "REAL"
|
|
472
|
+
text = str(value).strip()
|
|
473
|
+
if _INTEGER.match(text):
|
|
474
|
+
return "INTEGER"
|
|
475
|
+
if _REAL.match(text):
|
|
476
|
+
return "REAL"
|
|
477
|
+
return "TEXT"
|
|
478
|
+
|
|
479
|
+
|
|
480
|
+
def infer_types(rows: Sequence[Sequence[object]], count: int) -> list[str]:
|
|
481
|
+
"""Pick INTEGER, REAL or TEXT for each of the *count* columns from the values seen."""
|
|
482
|
+
rank = {"INTEGER": 0, "REAL": 1, "TEXT": 2}
|
|
483
|
+
types = ["INTEGER"] * count
|
|
484
|
+
seen = [False] * count
|
|
485
|
+
for row in rows:
|
|
486
|
+
for index in range(count):
|
|
487
|
+
value = row[index] if index < len(row) else None
|
|
488
|
+
if value is None:
|
|
489
|
+
continue
|
|
490
|
+
seen[index] = True
|
|
491
|
+
kind = _classify(value)
|
|
492
|
+
if rank[kind] > rank[types[index]]:
|
|
493
|
+
types[index] = kind
|
|
494
|
+
return [kind if was_seen else "TEXT" for kind, was_seen in zip(types, seen, strict=True)]
|
|
495
|
+
|
|
496
|
+
|
|
497
|
+
def coerce_rows(rows: Sequence[Sequence[object]], types: Sequence[str]) -> list[list[object]]:
|
|
498
|
+
"""Convert textual values to int/float according to *types*."""
|
|
499
|
+
converters = {"INTEGER": int, "REAL": float}
|
|
500
|
+
coerced = []
|
|
501
|
+
for row in rows:
|
|
502
|
+
values: list[object] = []
|
|
503
|
+
for value, kind in zip(row, types, strict=False):
|
|
504
|
+
if isinstance(value, str) and kind in converters:
|
|
505
|
+
value = converters[kind](value.strip())
|
|
506
|
+
values.append(value)
|
|
507
|
+
coerced.append(values)
|
|
508
|
+
return coerced
|