thunderduck-sqlalchemy 0.1.0__py3-none-any.whl
This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
- thunderduck_sqlalchemy/__init__.py +13 -0
- thunderduck_sqlalchemy/catalog.py +76 -0
- thunderduck_sqlalchemy/client.py +216 -0
- thunderduck_sqlalchemy/dbapi.py +306 -0
- thunderduck_sqlalchemy/dialect.py +242 -0
- thunderduck_sqlalchemy/exceptions.py +48 -0
- thunderduck_sqlalchemy/py.typed +0 -0
- thunderduck_sqlalchemy/superset_spec.py +124 -0
- thunderduck_sqlalchemy/types.py +232 -0
- thunderduck_sqlalchemy-0.1.0.dist-info/METADATA +153 -0
- thunderduck_sqlalchemy-0.1.0.dist-info/RECORD +13 -0
- thunderduck_sqlalchemy-0.1.0.dist-info/WHEEL +4 -0
- thunderduck_sqlalchemy-0.1.0.dist-info/entry_points.txt +7 -0
|
@@ -0,0 +1,242 @@
|
|
|
1
|
+
"""The SQLAlchemy dialect.
|
|
2
|
+
|
|
3
|
+
## Connection URL
|
|
4
|
+
|
|
5
|
+
thunderduck://:<tdk_token>@api.thunderduck.io/
|
|
6
|
+
thunderduck://:<tdk_token>@console-api.thunderduck.svc.cluster.local:8000/?ssl=false
|
|
7
|
+
|
|
8
|
+
The token is the URL *password* (or `?token=`). It is never interpolated into
|
|
9
|
+
a URL that could be logged. `?ssl=false` exists for in-cluster use, where
|
|
10
|
+
talking to the console-api Service directly avoids a round trip through the
|
|
11
|
+
public tunnel. `?poll_ms=` and `?timeout_ms=` mirror the JDBC driver's
|
|
12
|
+
parameter names.
|
|
13
|
+
|
|
14
|
+
## Reflection without SQL
|
|
15
|
+
|
|
16
|
+
`get_schema_names` and friends read console-api's `/catalog/schema` tree
|
|
17
|
+
rather than issuing `information_schema` queries -- issuing SQL here would
|
|
18
|
+
start a Kubernetes Job per reflection call. The tree is fetched once per
|
|
19
|
+
DBAPI connection and memoised on it.
|
|
20
|
+
|
|
21
|
+
`/catalog/tables` would give richer metadata (row counts, nullability) but it
|
|
22
|
+
is a cache that stays empty until a metadata refresh has run, so reflection
|
|
23
|
+
would silently return nothing on a fresh deployment. `/catalog/schema` is
|
|
24
|
+
live, and it is what the JDBC driver uses.
|
|
25
|
+
|
|
26
|
+
## Why `do_ping` is a lie
|
|
27
|
+
|
|
28
|
+
`DefaultDialect.do_ping` issues `SELECT 1`. Here that submits a query, starts
|
|
29
|
+
a Kubernetes Job and waits seconds for it -- on every pooled connection
|
|
30
|
+
checkout. A thunderduck connection holds no server-side state, so it cannot
|
|
31
|
+
go stale: reporting healthy without a round trip is both cheaper and more
|
|
32
|
+
accurate. Users should also pass `poolclass=NullPool` (see README).
|
|
33
|
+
"""
|
|
34
|
+
|
|
35
|
+
from __future__ import annotations
|
|
36
|
+
|
|
37
|
+
from typing import Any
|
|
38
|
+
|
|
39
|
+
from sqlalchemy.engine import default
|
|
40
|
+
from sqlalchemy.sql import compiler
|
|
41
|
+
|
|
42
|
+
from . import dbapi as _dbapi
|
|
43
|
+
from . import exceptions as exc
|
|
44
|
+
from .catalog import flatten, split_schema
|
|
45
|
+
from .types import sqlalchemy_type
|
|
46
|
+
|
|
47
|
+
#: Attribute name used to memoise the flattened catalog on a DBAPI connection.
|
|
48
|
+
_CATALOG_CACHE_ATTR = "_thunderduck_catalog_cache"
|
|
49
|
+
|
|
50
|
+
|
|
51
|
+
class ThunderduckIdentifierPreparer(compiler.IdentifierPreparer):
|
|
52
|
+
"""Quotes a composite schema segment-by-segment.
|
|
53
|
+
|
|
54
|
+
thunderduck schemas are dotted paths (`lego.public`), because SQLAlchemy
|
|
55
|
+
only models two name levels while thunderduck has up to three. Quoting the
|
|
56
|
+
whole thing would emit `"lego.public"` -- a single identifier containing a
|
|
57
|
+
dot, which resolves to nothing. Splitting first emits
|
|
58
|
+
`"lego"."public"`, which is what DuckDB (and console-api's sqlglot
|
|
59
|
+
rewriter) expects.
|
|
60
|
+
"""
|
|
61
|
+
|
|
62
|
+
def quote_schema(self, schema: str, force: Any = None) -> str:
|
|
63
|
+
segments = split_schema(schema)
|
|
64
|
+
return ".".join(self.quote_identifier(segment) for segment in segments)
|
|
65
|
+
|
|
66
|
+
def _requires_quotes(self, value: str) -> bool:
|
|
67
|
+
"""Quote every identifier, not just the ones SQLAlchemy thinks need it.
|
|
68
|
+
|
|
69
|
+
`quote_schema` above always quotes its segments, but the stock
|
|
70
|
+
`format_table` quotes the table name only *conditionally* -- so a plain
|
|
71
|
+
lowercase table came out as `"lego"."public".lego_sets`, mixing the two
|
|
72
|
+
styles. Forcing quotes everywhere makes the output uniform.
|
|
73
|
+
|
|
74
|
+
Scope: this is a preparer-wide hook, so it force-quotes every identifier
|
|
75
|
+
SQLAlchemy routes through `quote()` -- columns, labels, aliases, indexes
|
|
76
|
+
and constraints, not only schema and table names. Bind parameters are
|
|
77
|
+
unaffected; they never pass through `quote()`.
|
|
78
|
+
|
|
79
|
+
That breadth is safe because **DuckDB identifiers are case-insensitive
|
|
80
|
+
even when quoted** -- verified against duckdb: for a table created as
|
|
81
|
+
`MyTable`, all of `mytable`, `"mytable"` and `"MYTABLE"` resolve to it.
|
|
82
|
+
Unlike Postgres, quoting in DuckDB therefore changes only the rendered
|
|
83
|
+
SQL, never resolution. It also means reserved words and mixed-case names
|
|
84
|
+
need no special handling.
|
|
85
|
+
|
|
86
|
+
Server-side resolution is likewise unaffected: console-api's rewriter
|
|
87
|
+
matches on sqlglot's `table.catalog`/`.db`/`.name`, which return
|
|
88
|
+
unquoted text either way.
|
|
89
|
+
"""
|
|
90
|
+
return True
|
|
91
|
+
|
|
92
|
+
|
|
93
|
+
class ThunderduckDialect(default.DefaultDialect):
|
|
94
|
+
name = "thunderduck"
|
|
95
|
+
driver = "rest"
|
|
96
|
+
|
|
97
|
+
preparer = ThunderduckIdentifierPreparer
|
|
98
|
+
|
|
99
|
+
# Every statement is an HTTP round trip plus a Kubernetes Job, so there is
|
|
100
|
+
# nothing to gain from caching compiled forms, and `NullType` columns make
|
|
101
|
+
# cache keys unreliable.
|
|
102
|
+
supports_statement_cache = False
|
|
103
|
+
|
|
104
|
+
supports_native_boolean = True
|
|
105
|
+
supports_native_decimal = True
|
|
106
|
+
supports_sane_rowcount = False
|
|
107
|
+
supports_sane_multi_rowcount = False
|
|
108
|
+
postfetch_lastrowid = False
|
|
109
|
+
supports_statement_hints = False
|
|
110
|
+
default_paramstyle = "pyformat"
|
|
111
|
+
# DuckDB folds unquoted identifiers to lower case and is case-insensitive.
|
|
112
|
+
requires_name_normalize = False
|
|
113
|
+
max_identifier_length = 255
|
|
114
|
+
|
|
115
|
+
@classmethod
|
|
116
|
+
def import_dbapi(cls) -> Any:
|
|
117
|
+
return _dbapi
|
|
118
|
+
|
|
119
|
+
# SQLAlchemy 1.4 compatibility: it looks for `dbapi`, 2.0 for `import_dbapi`.
|
|
120
|
+
@classmethod
|
|
121
|
+
def dbapi(cls) -> Any:
|
|
122
|
+
return _dbapi
|
|
123
|
+
|
|
124
|
+
def create_connect_args(self, url: Any) -> tuple[list, dict]:
|
|
125
|
+
query = dict(url.query or {})
|
|
126
|
+
token = url.password or query.pop("token", None)
|
|
127
|
+
if not token:
|
|
128
|
+
raise exc.InterfaceError(
|
|
129
|
+
"no API token in the connection URL -- use "
|
|
130
|
+
"thunderduck://:<tdk_token>@host/ or ?token=<tdk_token>"
|
|
131
|
+
)
|
|
132
|
+
kwargs: dict[str, Any] = {
|
|
133
|
+
"host": url.host,
|
|
134
|
+
"token": token,
|
|
135
|
+
"use_ssl": str(query.pop("ssl", "true")).lower() not in ("false", "0", "no"),
|
|
136
|
+
}
|
|
137
|
+
if url.port:
|
|
138
|
+
kwargs["port"] = int(url.port)
|
|
139
|
+
poll_ms = query.pop("poll_ms", None)
|
|
140
|
+
if poll_ms is not None:
|
|
141
|
+
kwargs["poll_interval"] = int(poll_ms) / 1000.0
|
|
142
|
+
timeout_ms = query.pop("timeout_ms", None)
|
|
143
|
+
if timeout_ms is not None:
|
|
144
|
+
kwargs["timeout"] = int(timeout_ms) / 1000.0
|
|
145
|
+
arraysize = query.pop("arraysize", None)
|
|
146
|
+
if arraysize is not None:
|
|
147
|
+
kwargs["arraysize"] = int(arraysize)
|
|
148
|
+
return [], kwargs
|
|
149
|
+
|
|
150
|
+
def do_rollback(self, dbapi_connection: Any) -> None:
|
|
151
|
+
"""No transactions to roll back (read-only)."""
|
|
152
|
+
|
|
153
|
+
def do_commit(self, dbapi_connection: Any) -> None:
|
|
154
|
+
"""No transactions to commit (read-only)."""
|
|
155
|
+
|
|
156
|
+
def do_ping(self, dbapi_connection: Any) -> bool:
|
|
157
|
+
"""Always healthy. See the module docstring for why this issues no SQL."""
|
|
158
|
+
return True
|
|
159
|
+
|
|
160
|
+
# ---- reflection ----
|
|
161
|
+
|
|
162
|
+
def _tables(self, connection: Any) -> list:
|
|
163
|
+
"""The flattened catalog for this connection, fetched at most once."""
|
|
164
|
+
raw = getattr(connection, "connection", connection)
|
|
165
|
+
raw = getattr(raw, "dbapi_connection", raw)
|
|
166
|
+
cached = getattr(raw, _CATALOG_CACHE_ATTR, None)
|
|
167
|
+
if cached is None:
|
|
168
|
+
cached = flatten(raw.client.catalog_schema())
|
|
169
|
+
setattr(raw, _CATALOG_CACHE_ATTR, cached)
|
|
170
|
+
return cached
|
|
171
|
+
|
|
172
|
+
def get_schema_names(self, connection: Any, **kw: Any) -> list[str]:
|
|
173
|
+
return sorted({ref.schema for ref in self._tables(connection)})
|
|
174
|
+
|
|
175
|
+
def get_table_names(self, connection: Any, schema: str | None = None, **kw: Any) -> list[str]:
|
|
176
|
+
return [
|
|
177
|
+
ref.table
|
|
178
|
+
for ref in self._tables(connection)
|
|
179
|
+
if ref.schema == schema and ref.kind != "view"
|
|
180
|
+
]
|
|
181
|
+
|
|
182
|
+
def get_view_names(self, connection: Any, schema: str | None = None, **kw: Any) -> list[str]:
|
|
183
|
+
return [
|
|
184
|
+
ref.table
|
|
185
|
+
for ref in self._tables(connection)
|
|
186
|
+
if ref.schema == schema and ref.kind == "view"
|
|
187
|
+
]
|
|
188
|
+
|
|
189
|
+
def get_columns(
|
|
190
|
+
self, connection: Any, table_name: str, schema: str | None = None, **kw: Any
|
|
191
|
+
) -> list[dict]:
|
|
192
|
+
ref = self._find(connection, table_name, schema)
|
|
193
|
+
if ref is None:
|
|
194
|
+
return []
|
|
195
|
+
return [
|
|
196
|
+
{
|
|
197
|
+
"name": column["name"],
|
|
198
|
+
"type": sqlalchemy_type(column["type"]),
|
|
199
|
+
# /catalog/schema carries no nullability, so report the
|
|
200
|
+
# permissive default rather than inventing a constraint.
|
|
201
|
+
"nullable": True,
|
|
202
|
+
"default": None,
|
|
203
|
+
"autoincrement": False,
|
|
204
|
+
}
|
|
205
|
+
for column in ref.columns
|
|
206
|
+
]
|
|
207
|
+
|
|
208
|
+
def has_table(
|
|
209
|
+
self, connection: Any, table_name: str, schema: str | None = None, **kw: Any
|
|
210
|
+
) -> bool:
|
|
211
|
+
return self._find(connection, table_name, schema) is not None
|
|
212
|
+
|
|
213
|
+
def get_pk_constraint(
|
|
214
|
+
self, connection: Any, table_name: str, schema: str | None = None, **kw: Any
|
|
215
|
+
) -> dict:
|
|
216
|
+
return {"constrained_columns": [], "name": None}
|
|
217
|
+
|
|
218
|
+
def get_foreign_keys(
|
|
219
|
+
self, connection: Any, table_name: str, schema: str | None = None, **kw: Any
|
|
220
|
+
) -> list:
|
|
221
|
+
return []
|
|
222
|
+
|
|
223
|
+
def get_indexes(
|
|
224
|
+
self, connection: Any, table_name: str, schema: str | None = None, **kw: Any
|
|
225
|
+
) -> list:
|
|
226
|
+
return []
|
|
227
|
+
|
|
228
|
+
def get_table_comment(
|
|
229
|
+
self, connection: Any, table_name: str, schema: str | None = None, **kw: Any
|
|
230
|
+
) -> dict:
|
|
231
|
+
return {"text": None}
|
|
232
|
+
|
|
233
|
+
def _find(self, connection: Any, table_name: str, schema: str | None):
|
|
234
|
+
for ref in self._tables(connection):
|
|
235
|
+
if ref.table == table_name and (schema is None or ref.schema == schema):
|
|
236
|
+
return ref
|
|
237
|
+
return None
|
|
238
|
+
|
|
239
|
+
|
|
240
|
+
#: SQLAlchemy resolves `thunderduck://` and `thunderduck+rest://` here via the
|
|
241
|
+
#: entry points declared in pyproject.toml.
|
|
242
|
+
dialect = ThunderduckDialect
|
|
@@ -0,0 +1,48 @@
|
|
|
1
|
+
"""PEP 249 exception hierarchy.
|
|
2
|
+
|
|
3
|
+
Defined here rather than in `dbapi.py` so every module (client, types,
|
|
4
|
+
dialect) can raise the right class without importing the DBAPI layer, which
|
|
5
|
+
would be a circular import.
|
|
6
|
+
"""
|
|
7
|
+
|
|
8
|
+
|
|
9
|
+
class Error(Exception):
|
|
10
|
+
"""Base class for every error this driver raises."""
|
|
11
|
+
|
|
12
|
+
|
|
13
|
+
class InterfaceError(Error):
|
|
14
|
+
"""A problem with the driver or the connection itself, not the database.
|
|
15
|
+
|
|
16
|
+
Used for bad connection URLs, missing tokens and auth failures.
|
|
17
|
+
"""
|
|
18
|
+
|
|
19
|
+
|
|
20
|
+
class DatabaseError(Error):
|
|
21
|
+
"""A problem reported by thunderduck about the request."""
|
|
22
|
+
|
|
23
|
+
|
|
24
|
+
class DataError(DatabaseError):
|
|
25
|
+
"""A value could not be represented in SQL (e.g. NaN, a NUL byte)."""
|
|
26
|
+
|
|
27
|
+
|
|
28
|
+
class OperationalError(DatabaseError):
|
|
29
|
+
"""A run-time failure outside the caller's control.
|
|
30
|
+
|
|
31
|
+
Cancelled queries, poll timeouts and expired result sets.
|
|
32
|
+
"""
|
|
33
|
+
|
|
34
|
+
|
|
35
|
+
class IntegrityError(DatabaseError):
|
|
36
|
+
"""Relational integrity was violated. Not raised today (read-only)."""
|
|
37
|
+
|
|
38
|
+
|
|
39
|
+
class InternalError(DatabaseError):
|
|
40
|
+
"""The driver reached a state that should be impossible."""
|
|
41
|
+
|
|
42
|
+
|
|
43
|
+
class ProgrammingError(DatabaseError):
|
|
44
|
+
"""The caller did something wrong: bad SQL, or an unbindable parameter."""
|
|
45
|
+
|
|
46
|
+
|
|
47
|
+
class NotSupportedError(DatabaseError):
|
|
48
|
+
"""A DBAPI feature thunderduck does not have (transactions, DML)."""
|
|
File without changes
|
|
@@ -0,0 +1,124 @@
|
|
|
1
|
+
"""Superset DB engine spec for thunderduck.
|
|
2
|
+
|
|
3
|
+
Registered through the `superset.db_engine_specs` entry point in
|
|
4
|
+
pyproject.toml, so simply pip-installing this package into a Superset image is
|
|
5
|
+
enough -- no Superset configuration required.
|
|
6
|
+
|
|
7
|
+
## Why it subclasses DuckDB's spec
|
|
8
|
+
|
|
9
|
+
thunderduck executes DuckDB SQL. Superset's `DuckDBEngineSpec` already
|
|
10
|
+
encodes the right time-grain expressions (`DATE_TRUNC` forms), LIMIT handling
|
|
11
|
+
and column-type conversion for that SQL, so inheriting is both less code and
|
|
12
|
+
more correct than re-deriving them.
|
|
13
|
+
|
|
14
|
+
## Query cancellation
|
|
15
|
+
|
|
16
|
+
The Java JDBC driver's `Statement.cancel()` is a no-op, so BI tools cannot
|
|
17
|
+
stop a running thunderduck query through it. Here the cursor knows its
|
|
18
|
+
execution uuid, and console-api exposes `POST /executions/{id}/cancel`, so
|
|
19
|
+
Superset's stop button really does kill the Kubernetes Job.
|
|
20
|
+
|
|
21
|
+
Superset is deliberately NOT a dependency of this package -- it supplies
|
|
22
|
+
itself at runtime. This module therefore imports `superset` at module level
|
|
23
|
+
and is only ever imported by Superset resolving its own entry point. It must
|
|
24
|
+
never wrap that import in a try/except: a silently degraded engine spec would
|
|
25
|
+
give confusing behaviour instead of a clear failure.
|
|
26
|
+
"""
|
|
27
|
+
|
|
28
|
+
from __future__ import annotations
|
|
29
|
+
|
|
30
|
+
from typing import Any
|
|
31
|
+
|
|
32
|
+
from superset.db_engine_specs.base import BaseEngineSpec
|
|
33
|
+
from superset.db_engine_specs.duckdb import DuckDBEngineSpec
|
|
34
|
+
|
|
35
|
+
|
|
36
|
+
class ThunderduckEngineSpec(DuckDBEngineSpec):
|
|
37
|
+
engine = "thunderduck"
|
|
38
|
+
engine_name = "Thunderduck"
|
|
39
|
+
default_driver = "rest"
|
|
40
|
+
|
|
41
|
+
# No `engine_aliases`. Superset matches a spec on SQLAlchemy's
|
|
42
|
+
# `url.get_backend_name()`, which is `"thunderduck"` for BOTH
|
|
43
|
+
# `thunderduck://` and `thunderduck+rest://` (verified) -- the dotted
|
|
44
|
+
# `thunderduck.rest` form exists only as an entry-point KEY in
|
|
45
|
+
# pyproject.toml, which is a different namespace and never appears in a URL.
|
|
46
|
+
# An alias for it would therefore be dead config, so `engine` alone covers
|
|
47
|
+
# both driver spellings.
|
|
48
|
+
|
|
49
|
+
# Read-only service: no file uploads, no CTAS/CVAS.
|
|
50
|
+
supports_file_upload = False
|
|
51
|
+
|
|
52
|
+
# Ordinary SELECT capabilities, stated explicitly rather than inherited so
|
|
53
|
+
# a change in a future Superset default is visible here.
|
|
54
|
+
allows_alias_in_select = True
|
|
55
|
+
allows_subqueries = True
|
|
56
|
+
|
|
57
|
+
sqlalchemy_uri_placeholder = "thunderduck://:{api_token}@api.thunderduck.io/"
|
|
58
|
+
|
|
59
|
+
@staticmethod
|
|
60
|
+
def get_extra_params(database: Any, source: Any = None) -> dict[str, Any]:
|
|
61
|
+
"""Skip DuckDB's `connect_args["config"]` injection.
|
|
62
|
+
|
|
63
|
+
`DuckDBEngineSpec.get_extra_params` appends a user-agent by injecting
|
|
64
|
+
`engine_params.connect_args.config`, which is a DuckDB *file* setting.
|
|
65
|
+
Our `connect()` has an explicit keyword-only signature and rejects an
|
|
66
|
+
unknown `config` kwarg with TypeError -- so inheriting this makes it
|
|
67
|
+
impossible to add a thunderduck database in Superset AT ALL. Reproduced
|
|
68
|
+
live in a 6.1.0 container: both Test Connection and database creation
|
|
69
|
+
fail. Delegating to BaseEngineSpec keeps the useful behaviour and drops
|
|
70
|
+
the DuckDB-specific part.
|
|
71
|
+
"""
|
|
72
|
+
return BaseEngineSpec.get_extra_params(database)
|
|
73
|
+
|
|
74
|
+
@classmethod
|
|
75
|
+
def fetch_data(cls, cursor: Any, limit: int | None = None) -> list[Any]:
|
|
76
|
+
"""Skip DuckDB's cursor.description assignment.
|
|
77
|
+
|
|
78
|
+
`DuckDBEngineSpec.fetch_data` works around duckdb-engine's issue #1322
|
|
79
|
+
(cursor.description becomes None after fetchall) by capturing and
|
|
80
|
+
reassigning it: `cursor.description = description`. Our Cursor.description
|
|
81
|
+
is a read-only property, so that raises AttributeError *after the query
|
|
82
|
+
has already run as a Kubernetes Job* -- the customer is charged and gets
|
|
83
|
+
a 500 error. Delegating to BaseEngineSpec skips the DuckDB workaround
|
|
84
|
+
and works correctly here.
|
|
85
|
+
|
|
86
|
+
This exemplifies the inheritance rule: we want DuckDB SQL behaviour
|
|
87
|
+
(time expressions, type conversion, LIMIT handling), not duckdb-engine's
|
|
88
|
+
Python driver workarounds. Override driver-specific hacks; inherit SQL.
|
|
89
|
+
"""
|
|
90
|
+
return super(DuckDBEngineSpec, cls).fetch_data(cursor, limit)
|
|
91
|
+
|
|
92
|
+
# NOTE: `has_implicit_cancel` is deliberately NOT overridden -- it must stay
|
|
93
|
+
# False (BaseEngineSpec's default, which DuckDBEngineSpec also keeps).
|
|
94
|
+
#
|
|
95
|
+
# Returning True makes `superset/sql/execution/executor.py` return success
|
|
96
|
+
# from the stop-query path *immediately*, without ever calling
|
|
97
|
+
# `cancel_query` -- so the UI would report a query stopped while its
|
|
98
|
+
# Kubernetes Job kept running and kept costing money. The flag means "the
|
|
99
|
+
# live cursor cancels implicitly", which is not us: our cancel is an
|
|
100
|
+
# explicit REST call.
|
|
101
|
+
#
|
|
102
|
+
# Honest status of stop-button support: it does NOT work yet, and the reason
|
|
103
|
+
# is structural rather than a missing flag. Superset only records a cancel
|
|
104
|
+
# id when `has_query_id_before_execute` is False, and it does so *after*
|
|
105
|
+
# `execute()` returns; our `execute()` blocks until the query has finished,
|
|
106
|
+
# so by then there is nothing left to cancel. Making the stop button work
|
|
107
|
+
# needs a genuinely async execute path (submit, return, poll in
|
|
108
|
+
# `handle_cursor`) -- a real follow-up, not a tweak. The two hooks below are
|
|
109
|
+
# correct and ready for that day; cancellation already works today for
|
|
110
|
+
# Python callers via `cursor.cancel()` (verified live against a running
|
|
111
|
+
# query, which reached terminal status `cancelled`).
|
|
112
|
+
|
|
113
|
+
@classmethod
|
|
114
|
+
def get_cancel_query_id(cls, cursor: Any, query: Any) -> str | None:
|
|
115
|
+
"""thunderduck's own execution uuid identifies the run to cancel."""
|
|
116
|
+
return getattr(cursor, "execution_id", None)
|
|
117
|
+
|
|
118
|
+
@classmethod
|
|
119
|
+
def cancel_query(cls, cursor: Any, query: Any, cancel_query_id: str) -> bool:
|
|
120
|
+
try:
|
|
121
|
+
cursor.cancel()
|
|
122
|
+
except Exception: # noqa: BLE001 - Superset expects a bool, not a raise
|
|
123
|
+
return False
|
|
124
|
+
return True
|
|
@@ -0,0 +1,232 @@
|
|
|
1
|
+
"""DuckDB type names -> SQLAlchemy types, and Python values -> SQL literals.
|
|
2
|
+
|
|
3
|
+
## Two directions, two failure policies
|
|
4
|
+
|
|
5
|
+
Reading (`sqlalchemy_type`) is a *display* concern: an unrecognised type name
|
|
6
|
+
degrades to `NullType` so an exotic column never breaks a whole query.
|
|
7
|
+
|
|
8
|
+
Writing (`render_literal`) is a *security* boundary. console-api's
|
|
9
|
+
`POST /queries` accepts raw SQL only -- there is no server-side parameter
|
|
10
|
+
bind -- so this module renders literals into the statement. An unrecognised
|
|
11
|
+
Python type therefore raises: a silent `str()` fallback is precisely how a
|
|
12
|
+
wrong-type bug or an injected string reaches generated SQL.
|
|
13
|
+
"""
|
|
14
|
+
|
|
15
|
+
from __future__ import annotations
|
|
16
|
+
|
|
17
|
+
import datetime as dt
|
|
18
|
+
import math
|
|
19
|
+
import re
|
|
20
|
+
from decimal import Decimal
|
|
21
|
+
from typing import Any
|
|
22
|
+
|
|
23
|
+
from sqlalchemy import types as sqltypes
|
|
24
|
+
|
|
25
|
+
from . import exceptions as exc
|
|
26
|
+
|
|
27
|
+
# `Uuid` arrived in SQLAlchemy 2.0. This package supports 1.4 as well (Superset
|
|
28
|
+
# 6.1.0 pins sqlalchemy<2), so resolve it dynamically rather than importing it:
|
|
29
|
+
# on 1.4 a UUID column reflects as String, which is usable, instead of the
|
|
30
|
+
# module failing to import at all.
|
|
31
|
+
_UUID_TYPE: Any = getattr(sqltypes, "Uuid", sqltypes.String)
|
|
32
|
+
|
|
33
|
+
# NOTE: `/catalog/schema` reports the SOURCE system's type names, not DuckDB's.
|
|
34
|
+
# A PostgreSQL-backed catalog yields `character varying`, `double precision`,
|
|
35
|
+
# `jsonb` and friends -- verified against the live deployment, where a table's
|
|
36
|
+
# columns came back as `integer`, `character varying`, `character`. So this map
|
|
37
|
+
# has to cover both vocabularies; a miss is not cosmetic, because a column that
|
|
38
|
+
# degrades to NullType loses its filters and groupings in a BI tool.
|
|
39
|
+
_SIMPLE_TYPES: dict[str, Any] = {
|
|
40
|
+
"BOOLEAN": sqltypes.Boolean,
|
|
41
|
+
"BOOL": sqltypes.Boolean,
|
|
42
|
+
# --- PostgreSQL spellings ---
|
|
43
|
+
"CHARACTER VARYING": sqltypes.String,
|
|
44
|
+
"CHARACTER": sqltypes.String,
|
|
45
|
+
"BPCHAR": sqltypes.String,
|
|
46
|
+
"DOUBLE PRECISION": sqltypes.Float,
|
|
47
|
+
"JSONB": sqltypes.JSON,
|
|
48
|
+
"SMALLSERIAL": sqltypes.SmallInteger,
|
|
49
|
+
"SERIAL": sqltypes.Integer,
|
|
50
|
+
"BIGSERIAL": sqltypes.BigInteger,
|
|
51
|
+
"OID": sqltypes.Integer,
|
|
52
|
+
"MONEY": sqltypes.Numeric,
|
|
53
|
+
"INET": sqltypes.String,
|
|
54
|
+
"CIDR": sqltypes.String,
|
|
55
|
+
"MACADDR": sqltypes.String,
|
|
56
|
+
"XML": sqltypes.String,
|
|
57
|
+
"BIT": sqltypes.String,
|
|
58
|
+
"BIT VARYING": sqltypes.String,
|
|
59
|
+
"TINYINT": sqltypes.SmallInteger,
|
|
60
|
+
"SMALLINT": sqltypes.SmallInteger,
|
|
61
|
+
"INTEGER": sqltypes.Integer,
|
|
62
|
+
"INT": sqltypes.Integer,
|
|
63
|
+
"BIGINT": sqltypes.BigInteger,
|
|
64
|
+
"HUGEINT": sqltypes.BigInteger,
|
|
65
|
+
"UBIGINT": sqltypes.BigInteger,
|
|
66
|
+
"UINTEGER": sqltypes.Integer,
|
|
67
|
+
"USMALLINT": sqltypes.SmallInteger,
|
|
68
|
+
"UTINYINT": sqltypes.SmallInteger,
|
|
69
|
+
"FLOAT": sqltypes.Float,
|
|
70
|
+
"REAL": sqltypes.Float,
|
|
71
|
+
"DOUBLE": sqltypes.Float,
|
|
72
|
+
"VARCHAR": sqltypes.String,
|
|
73
|
+
"TEXT": sqltypes.String,
|
|
74
|
+
"STRING": sqltypes.String,
|
|
75
|
+
"DATE": sqltypes.Date,
|
|
76
|
+
"TIME": sqltypes.Time,
|
|
77
|
+
"BLOB": sqltypes.LargeBinary,
|
|
78
|
+
"BYTEA": sqltypes.LargeBinary,
|
|
79
|
+
"UUID": _UUID_TYPE,
|
|
80
|
+
"JSON": sqltypes.JSON,
|
|
81
|
+
"INTERVAL": sqltypes.Interval,
|
|
82
|
+
}
|
|
83
|
+
|
|
84
|
+
_DECIMAL_RE = re.compile(r"^(?:DECIMAL|NUMERIC)\s*\(\s*(\d+)\s*,\s*(\d+)\s*\)$")
|
|
85
|
+
|
|
86
|
+
|
|
87
|
+
def sqlalchemy_type(duckdb_type: str) -> Any:
|
|
88
|
+
"""Map one `column_types` entry onto a SQLAlchemy type instance."""
|
|
89
|
+
name = (duckdb_type or "").strip().upper()
|
|
90
|
+
|
|
91
|
+
if name.endswith("[]"):
|
|
92
|
+
return sqltypes.ARRAY(sqlalchemy_type(name[:-2]))
|
|
93
|
+
|
|
94
|
+
if name.startswith("TIMESTAMP"):
|
|
95
|
+
# TIMESTAMP, TIMESTAMP WITH TIME ZONE, TIMESTAMPTZ, TIMESTAMP_NS ...
|
|
96
|
+
timezone = "WITH TIME ZONE" in name or name.endswith("TZ")
|
|
97
|
+
return sqltypes.DateTime(timezone=timezone)
|
|
98
|
+
|
|
99
|
+
# Must come AFTER the TIMESTAMP branch -- "TIMESTAMP..." also starts with
|
|
100
|
+
# "TIME". Covers PostgreSQL's `time without time zone` / `time with time
|
|
101
|
+
# zone` as well as a bare `TIME`.
|
|
102
|
+
if name.startswith("TIME"):
|
|
103
|
+
return sqltypes.Time(timezone="WITH TIME ZONE" in name)
|
|
104
|
+
|
|
105
|
+
decimal_match = _DECIMAL_RE.match(name)
|
|
106
|
+
if decimal_match:
|
|
107
|
+
precision, scale = decimal_match.groups()
|
|
108
|
+
return sqltypes.Numeric(precision=int(precision), scale=int(scale))
|
|
109
|
+
if name in ("DECIMAL", "NUMERIC"):
|
|
110
|
+
return sqltypes.Numeric()
|
|
111
|
+
|
|
112
|
+
simple = _SIMPLE_TYPES.get(name)
|
|
113
|
+
if simple is not None:
|
|
114
|
+
return simple()
|
|
115
|
+
|
|
116
|
+
# A bare `ARRAY` with no element type -- PostgreSQL reports this for array
|
|
117
|
+
# columns. Keep it an ARRAY so a consumer still knows the shape, with an
|
|
118
|
+
# unknown item type rather than losing the column entirely.
|
|
119
|
+
if name == "ARRAY":
|
|
120
|
+
return sqltypes.ARRAY(sqltypes.NullType())
|
|
121
|
+
|
|
122
|
+
# VARCHAR(255) and friends: strip the parameters and retry once.
|
|
123
|
+
base = name.split("(", 1)[0].strip()
|
|
124
|
+
simple = _SIMPLE_TYPES.get(base)
|
|
125
|
+
if simple is not None:
|
|
126
|
+
return simple()
|
|
127
|
+
|
|
128
|
+
return sqltypes.NullType()
|
|
129
|
+
|
|
130
|
+
|
|
131
|
+
def render_literal(value: Any) -> str:
|
|
132
|
+
"""Render one Python value as a DuckDB SQL literal.
|
|
133
|
+
|
|
134
|
+
Strict by design: anything not explicitly handled raises.
|
|
135
|
+
"""
|
|
136
|
+
if value is None:
|
|
137
|
+
return "NULL"
|
|
138
|
+
# bool before int -- bool is a subclass of int in Python.
|
|
139
|
+
if isinstance(value, bool):
|
|
140
|
+
return "TRUE" if value else "FALSE"
|
|
141
|
+
if isinstance(value, int):
|
|
142
|
+
return str(value)
|
|
143
|
+
if isinstance(value, float):
|
|
144
|
+
if not math.isfinite(value):
|
|
145
|
+
raise exc.DataError(f"cannot render non-finite float {value!r} as SQL")
|
|
146
|
+
return repr(value)
|
|
147
|
+
if isinstance(value, Decimal):
|
|
148
|
+
if not value.is_finite():
|
|
149
|
+
raise exc.DataError(f"cannot render non-finite decimal {value!r} as SQL")
|
|
150
|
+
return str(value)
|
|
151
|
+
if isinstance(value, str):
|
|
152
|
+
return _quote_string(value)
|
|
153
|
+
if isinstance(value, (bytes, bytearray, memoryview)):
|
|
154
|
+
return f"unhex('{bytes(value).hex()}')::BLOB"
|
|
155
|
+
if isinstance(value, dt.datetime):
|
|
156
|
+
return f"'{value.isoformat(sep=' ')}'::TIMESTAMP"
|
|
157
|
+
if isinstance(value, dt.date):
|
|
158
|
+
return f"'{value.isoformat()}'::DATE"
|
|
159
|
+
if isinstance(value, dt.time):
|
|
160
|
+
return f"'{value.isoformat()}'::TIME"
|
|
161
|
+
if isinstance(value, (list, tuple, set, frozenset)):
|
|
162
|
+
items = ", ".join(render_literal(v) for v in value)
|
|
163
|
+
return f"[{items}]"
|
|
164
|
+
raise exc.ProgrammingError(
|
|
165
|
+
f"cannot bind a value of type {type(value).__name__} -- "
|
|
166
|
+
"thunderduck renders parameters client-side and only supports "
|
|
167
|
+
"None, bool, int, float, Decimal, str, bytes, date/time/datetime and sequences"
|
|
168
|
+
)
|
|
169
|
+
|
|
170
|
+
|
|
171
|
+
def _quote_string(value: str) -> str:
|
|
172
|
+
if "\x00" in value:
|
|
173
|
+
raise exc.DataError("cannot bind a string containing a NUL byte")
|
|
174
|
+
# DuckDB does not process backslash escapes in standard single-quoted
|
|
175
|
+
# strings, so doubling the quote character is the whole escape.
|
|
176
|
+
return "'" + value.replace("'", "''") + "'"
|
|
177
|
+
|
|
178
|
+
|
|
179
|
+
def render_sql(sql: str, params: Any) -> str:
|
|
180
|
+
"""Substitute `pyformat` parameters into `sql`.
|
|
181
|
+
|
|
182
|
+
With no params the statement is returned untouched -- important, because
|
|
183
|
+
literal `%` signs are common in SQL (`like '%x%'`) and must not be treated
|
|
184
|
+
as placeholders.
|
|
185
|
+
"""
|
|
186
|
+
if params is None or (hasattr(params, "__len__") and len(params) == 0 and not sql_needs(sql)):
|
|
187
|
+
return sql
|
|
188
|
+
if isinstance(params, dict):
|
|
189
|
+
try:
|
|
190
|
+
return sql % {k: render_literal(v) for k, v in params.items()}
|
|
191
|
+
except KeyError as err:
|
|
192
|
+
raise exc.ProgrammingError(f"missing bind parameter {err.args[0]!r}") from None
|
|
193
|
+
except (TypeError, ValueError) as err:
|
|
194
|
+
raise exc.ProgrammingError(_placeholder_hint(err)) from None
|
|
195
|
+
if isinstance(params, (list, tuple)):
|
|
196
|
+
rendered = tuple(render_literal(v) for v in params)
|
|
197
|
+
try:
|
|
198
|
+
return sql % rendered
|
|
199
|
+
except (TypeError, ValueError) as err:
|
|
200
|
+
raise exc.ProgrammingError(_placeholder_hint(err, len(rendered))) from None
|
|
201
|
+
raise exc.ProgrammingError(
|
|
202
|
+
f"parameters must be a mapping or a sequence, got {type(params).__name__}"
|
|
203
|
+
)
|
|
204
|
+
|
|
205
|
+
|
|
206
|
+
def sql_needs(sql: str) -> bool:
|
|
207
|
+
"""True if the statement contains a pyformat placeholder."""
|
|
208
|
+
return "%(" in sql or "%s" in sql
|
|
209
|
+
|
|
210
|
+
|
|
211
|
+
def _placeholder_hint(err: Exception, supplied: int | None = None) -> str:
|
|
212
|
+
"""Turn Python's opaque format error into actionable advice.
|
|
213
|
+
|
|
214
|
+
Python does NOT distinguish the two ways this fails: both "too few
|
|
215
|
+
parameters supplied" and "the statement contains an unescaped literal
|
|
216
|
+
percent sign" surface as
|
|
217
|
+
`TypeError: not enough arguments for format string` (verified on 3.10-3.13
|
|
218
|
+
-- an earlier version of this code wrongly assumed ValueError and reported
|
|
219
|
+
every failure as a count mismatch). Since we cannot tell them apart, the
|
|
220
|
+
message names both causes rather than guessing.
|
|
221
|
+
|
|
222
|
+
`paramstyle` is `pyformat`, so -- exactly as with psycopg2 -- a literal
|
|
223
|
+
percent sign must be written `%%` when parameters are supplied. SQLAlchemy
|
|
224
|
+
does this automatically when it compiles; hand-written raw SQL does not.
|
|
225
|
+
"""
|
|
226
|
+
counted = f" {supplied} parameter(s) were supplied." if supplied is not None else ""
|
|
227
|
+
return (
|
|
228
|
+
f"could not bind parameters ({err}).{counted} Either the number of "
|
|
229
|
+
"placeholders does not match the parameters, or the statement contains "
|
|
230
|
+
"a literal percent sign that must be escaped as '%%' "
|
|
231
|
+
"(e.g. LIKE '%%abc%%') -- thunderduck uses the 'pyformat' paramstyle"
|
|
232
|
+
)
|