queryapigate 0.5.0__py3-none-any.whl

This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
@@ -0,0 +1,795 @@
1
+ """One runner per database type: connect (or borrow a pooled connection), run one statement, return (columns, rows).
2
+
3
+ Runners fetch ``limit + 1`` rows: the extra row is how the caller learns whether another page exists.
4
+ Database drivers are imported lazily so only the ones you actually use need to be installed.
5
+
6
+ Each network database is a small ``_Driver`` subclass. ``_make_runner`` turns it into the callable stored in
7
+ ``RUNNERS``, which either opens a connection for the single call or borrows one from the pool.
8
+ """
9
+ from __future__ import annotations # lets `str | None` below run on Python 3.9 too (annotations aren't evaluated)
10
+
11
+ import atexit
12
+ import logging
13
+ import math
14
+ import os
15
+ import sqlite3
16
+ import threading
17
+ import time
18
+ from pathlib import Path
19
+
20
+ from . import config, store
21
+ from .errors import ApiError
22
+ from .pool import Session, close_pooled_connections
23
+ from .sqltools import bind_parameters, is_paginated, paginate
24
+
25
+ log = logging.getLogger('queryapigate')
26
+
27
+
28
+ def _timed_out(timeout):
29
+ return ApiError(f'The query exceeded the time limit of {timeout:g} seconds and was cancelled', 504,
30
+ timeout=timeout)
31
+
32
+
33
+ def _millis(timeout):
34
+ return max(1, int(timeout * 1000))
35
+
36
+
37
+ def _connect_args(details, **renames):
38
+ """Pick the standard connection fields out of a stored connection, renaming keys per driver."""
39
+ args = {}
40
+ for key in ('host', 'port', 'user', 'password', 'database'):
41
+ if details.get(key) not in (None, ''):
42
+ args[renames.get(key, key)] = details[key]
43
+ return args
44
+
45
+
46
+ def _resolve_db_file(details, label):
47
+ """The local file path for a `database` connection field: resolved against QUERYAPIGATE_HOME, and required
48
+ to already exist - so a typo in the path 404s instead of silently opening (SQLite) or creating
49
+ (DuckDB) a fresh, empty database file at the wrong location."""
50
+ path = details.get('database')
51
+ if not path:
52
+ raise ApiError('Database file path not provided')
53
+ if not os.path.isabs(path):
54
+ path = os.path.join(config.home(), path)
55
+ if not os.path.isfile(path):
56
+ raise ApiError(f'{label} database file not found', 404)
57
+ return path
58
+
59
+
60
+ def _prepare(sql, params, style, limit, offset, dialect):
61
+ """Return (sql, args, sliced) - the statement to send, its bound arguments and whether the
62
+ database already applied the requested window (LIMIT/OFFSET) for us."""
63
+ window = limit + 1
64
+ if is_paginated(sql, dialect):
65
+ sql, sliced = paginate(sql, window, offset), True
66
+ else:
67
+ sliced = False
68
+ sql, args = bind_parameters(sql, params or {}, style, dialect)
69
+ return sql, args, sliced
70
+
71
+
72
+ def _fetch_page(cursor, sql, params, style, limit, offset, dialect):
73
+ sql, args, sliced = _prepare(sql, params, style, limit, offset, dialect)
74
+ if args is None:
75
+ cursor.execute(sql)
76
+ else:
77
+ cursor.execute(sql, args)
78
+ if not cursor.description:
79
+ return [], []
80
+ rows = cursor.fetchall() if sliced else cursor.fetchmany(offset + limit + 1)[offset:]
81
+ return [d[0] for d in cursor.description], rows
82
+
83
+
84
+ # --------------------------------------------------------------------------------------
85
+ # Streaming: no LIMIT/OFFSET, and never more than one _STREAM_BATCH of rows in memory at
86
+ # once, however large the full result is. `_stream_cursor` yields the column names first
87
+ # (a one-item "sentinel" yield, so the caller can prime the generator once to learn them
88
+ # before deciding anything about the response, then hand the rest of the generator - the
89
+ # actual rows - straight to the client without this module needing to know about HTTP at
90
+ # all) and then every row, one at a time, pulled from the database in _STREAM_BATCH-sized
91
+ # fetchmany() calls rather than a single fetchall(). A driver whose cursor already streams
92
+ # from the server without buffering the whole result client-side (an unbuffered MySQL
93
+ # cursor, a named/server-side PostgreSQL cursor, ClickHouse's execute_iter) additionally
94
+ # keeps *this process's* memory use flat regardless of result size; the others (SQLite,
95
+ # DuckDB, H2/JDBC) still bound it to one batch at a time even where the underlying engine
96
+ # or driver computes the full result before the first fetch - see runners.py driver
97
+ # docstrings and DATABASE_CONNECTION_CONFIGURATION.md for which is which.
98
+ # --------------------------------------------------------------------------------------
99
+
100
+ _STREAM_BATCH = 1000
101
+
102
+
103
+ def _stream_cursor(cursor, sql, params, style, dialect):
104
+ sql, args = bind_parameters(sql, params or {}, style, dialect)
105
+ if args is None:
106
+ cursor.execute(sql)
107
+ else:
108
+ cursor.execute(sql, args)
109
+ if not cursor.description:
110
+ yield ()
111
+ return
112
+ yield tuple(d[0] for d in cursor.description)
113
+ while True:
114
+ batch = cursor.fetchmany(_STREAM_BATCH)
115
+ if not batch:
116
+ return
117
+ yield from batch
118
+
119
+
120
+ # --------------------------------------------------------------------------------------
121
+ # Drivers
122
+ # --------------------------------------------------------------------------------------
123
+
124
+ class _Driver:
125
+ """How to connect to, check, reset, query and close one kind of database."""
126
+
127
+ DIALECT: str | None = None # the connection's `db` value; used to pick the right literal-quoting rules
128
+ PARAM_STYLE: str | None = None # this driver's bind_parameters() style; used by the default stream()
129
+
130
+ def connect(self, details, read_only):
131
+ raise NotImplementedError
132
+
133
+ def is_alive(self, session):
134
+ """Cheap check that a connection that sat idle is still usable."""
135
+ return True
136
+
137
+ def reset(self, session):
138
+ """Put a used connection back into a clean state: end its transaction so no stale snapshot lingers."""
139
+
140
+ def close(self, session):
141
+ session.conn.close()
142
+
143
+ def query(self, session, sql, params, limit, offset, read_only, timeout):
144
+ raise NotImplementedError
145
+
146
+ def stream(self, session, sql, params, timeout):
147
+ """Like query(), but for the whole result rather than one page: no LIMIT/OFFSET, and returns the
148
+ _stream_cursor() generator directly (columns first, then every row) instead of a (columns, rows)
149
+ pair, so a caller consuming it lazily never forces more than one batch into memory. Always called
150
+ with a connection opened read-only (streaming never writes - see engine.stream_sql), so unlike
151
+ query() there is no read_only branch or commit() to consider here.
152
+
153
+ The default implementation (a plain cursor(), fetchmany()-batched) is correct for any driver whose
154
+ cursor supports fetchmany, which is every driver here; MySQL, PostgreSQL and ClickHouse override it
155
+ to additionally avoid buffering the whole result on the client side - see their own stream().
156
+ """
157
+ return _stream_cursor(session.conn.cursor(), sql, params, self.PARAM_STYLE, self.DIALECT)
158
+
159
+
160
+ # MySQL reports 3024 (ER_QUERY_TIMEOUT), MariaDB 1969 (ER_STATEMENT_TIMEOUT)
161
+ _MYSQL_TIMEOUT_ERRNOS = (3024, 1969)
162
+
163
+
164
+ def _set_mysql_timeout(cursor, session, timeout):
165
+ """Shared by _MySQL.query() and .stream(): MySQL limits SELECT statements in milliseconds; MariaDB
166
+ uses a differently named variable in seconds. 0 removes a limit an earlier request left behind on
167
+ this pooled connection."""
168
+ import mysql.connector
169
+ if timeout == session.state.get('timeout'):
170
+ return
171
+ try:
172
+ cursor.execute(f'SET SESSION max_execution_time = {_millis(timeout) if timeout else 0}')
173
+ except mysql.connector.Error:
174
+ try:
175
+ cursor.execute(f'SET SESSION max_statement_time = {timeout:g}' if timeout
176
+ else 'SET SESSION max_statement_time = 0')
177
+ except mysql.connector.Error:
178
+ log.warning('This MySQL server supports neither max_execution_time nor '
179
+ 'max_statement_time; the query time limit is not enforced')
180
+ session.state['timeout'] = timeout
181
+
182
+
183
+ class _MySQL(_Driver):
184
+ DIALECT = 'mysql'
185
+ PARAM_STYLE = 'format'
186
+
187
+ def connect(self, details, read_only):
188
+ import mysql.connector
189
+ conn = mysql.connector.connect(connection_timeout=config.CONNECT_TIMEOUT, **_connect_args(details))
190
+ if read_only:
191
+ try:
192
+ cursor = conn.cursor()
193
+ cursor.execute('SET SESSION TRANSACTION READ ONLY')
194
+ cursor.close()
195
+ except Exception:
196
+ conn.close()
197
+ raise
198
+ return conn
199
+
200
+ def is_alive(self, session):
201
+ return session.conn.is_connected()
202
+
203
+ def reset(self, session):
204
+ session.conn.rollback()
205
+
206
+ def query(self, session, sql, params, limit, offset, read_only, timeout):
207
+ import mysql.connector
208
+ conn = session.conn
209
+ cursor = conn.cursor()
210
+ try:
211
+ _set_mysql_timeout(cursor, session, timeout)
212
+ try:
213
+ result = _fetch_page(cursor, sql, params, 'format', limit, offset, self.DIALECT)
214
+ except mysql.connector.Error as error:
215
+ if getattr(error, 'errno', None) in _MYSQL_TIMEOUT_ERRNOS:
216
+ raise _timed_out(timeout) from None
217
+ raise
218
+ if not read_only:
219
+ conn.commit()
220
+ return result
221
+ finally:
222
+ cursor.close()
223
+
224
+ def stream(self, session, sql, params, timeout):
225
+ """Unlike query(), uses an *unbuffered* cursor (mysql-connector-python's default cursor fetches and
226
+ buffers the entire result set into the client on execute() - fine for one page, defeats the point
227
+ of streaming for a large export) - see DATABASE_CONNECTION_CONFIGURATION.md#streaming-exports.
228
+ Verified end-to-end against a real server: 1M rows streamed over real HTTP with this process's own
229
+ RSS sampled throughout - flat at ~49MB the entire way, against ~344MB for the same query fetched
230
+ the ordinary (buffered) way."""
231
+ import mysql.connector
232
+ conn = session.conn
233
+ with conn.cursor() as setter:
234
+ _set_mysql_timeout(setter, session, timeout)
235
+ try:
236
+ yield from _stream_cursor(conn.cursor(buffered=False), sql, params, 'format', self.DIALECT)
237
+ except mysql.connector.Error as error:
238
+ if getattr(error, 'errno', None) in _MYSQL_TIMEOUT_ERRNOS:
239
+ raise _timed_out(timeout) from None
240
+ raise
241
+
242
+
243
+ class _Postgres(_Driver):
244
+ DIALECT = 'postgres'
245
+ PARAM_STYLE = 'format'
246
+
247
+ def connect(self, details, read_only):
248
+ import psycopg2
249
+ conn = psycopg2.connect(connect_timeout=config.CONNECT_TIMEOUT, **_connect_args(details, database='dbname'))
250
+ if read_only:
251
+ try:
252
+ conn.set_session(readonly=True)
253
+ except Exception:
254
+ conn.close()
255
+ raise
256
+ return conn
257
+
258
+ def is_alive(self, session):
259
+ if session.conn.closed:
260
+ return False
261
+ with session.conn.cursor() as cursor:
262
+ cursor.execute('SELECT 1')
263
+ return True
264
+
265
+ def reset(self, session):
266
+ session.conn.rollback()
267
+
268
+ def query(self, session, sql, params, limit, offset, read_only, timeout):
269
+ import psycopg2.errors
270
+ conn = session.conn
271
+ with conn.cursor() as cursor:
272
+ if timeout:
273
+ # LOCAL scopes the limit to this transaction, so nothing leaks to the connection's next user
274
+ cursor.execute(f'SET LOCAL statement_timeout = {_millis(timeout)}')
275
+ try:
276
+ result = _fetch_page(cursor, sql, params, 'format', limit, offset, self.DIALECT)
277
+ except psycopg2.errors.QueryCanceled:
278
+ raise _timed_out(timeout) from None
279
+ if not read_only:
280
+ conn.commit()
281
+ return result
282
+
283
+ def stream(self, session, sql, params, timeout):
284
+ """Unlike query(), uses a named (server-side) cursor - PostgreSQL's default cursor, like MySQL's,
285
+ fetches the whole result set into the client on execute() - see
286
+ DATABASE_CONNECTION_CONFIGURATION.md#streaming-exports. A named cursor needs an open transaction,
287
+ which every connection here already sits in outside autocommit mode; it is torn down along with
288
+ that transaction by reset() when the connection is released, same as the LOCAL statement_timeout.
289
+
290
+ Does not reuse _stream_cursor(): a named cursor's execute() is really a `DECLARE CURSOR ... FOR
291
+ <query>` under the hood, so it returns immediately without running the query at all - .description
292
+ stays None, and statement_timeout has nothing to cancel yet, until the *first fetch*, which is what
293
+ actually runs it server-side. So the first fetchmany() has to happen before .description is read,
294
+ the reverse of every other driver here - verified against a real, deliberately slow PostgreSQL
295
+ query on a named cursor (confirms both that .description is only populated after that first fetch,
296
+ and that statement_timeout does still correctly cancel it there). Also verified end-to-end like
297
+ MySQL above: 1M rows over real HTTP, RSS flat at ~55MB throughout, against ~564MB buffered."""
298
+ import psycopg2.errors
299
+ conn = session.conn
300
+ with conn.cursor() as setter:
301
+ if timeout:
302
+ setter.execute(f'SET LOCAL statement_timeout = {_millis(timeout)}')
303
+ cursor = conn.cursor(name='queryapigate_stream')
304
+ cursor.itersize = _STREAM_BATCH # rows fetched from the server per underlying fetchmany() call
305
+ sql, args = bind_parameters(sql, params or {}, 'format', self.DIALECT)
306
+ try:
307
+ cursor.execute(sql) if args is None else cursor.execute(sql, args)
308
+ first_batch = cursor.fetchmany(_STREAM_BATCH)
309
+ except psycopg2.errors.QueryCanceled:
310
+ raise _timed_out(timeout) from None
311
+ if not cursor.description:
312
+ yield ()
313
+ return
314
+ yield tuple(d[0] for d in cursor.description)
315
+ yield from first_batch
316
+ try:
317
+ while True:
318
+ batch = cursor.fetchmany(_STREAM_BATCH)
319
+ if not batch:
320
+ return
321
+ yield from batch
322
+ except psycopg2.errors.QueryCanceled:
323
+ raise _timed_out(timeout) from None
324
+
325
+
326
+ _CLICKHOUSE_TIMEOUT_EXCEEDED = 159
327
+
328
+
329
+ class _ClickHouse(_Driver):
330
+ DIALECT = 'clickhouse'
331
+
332
+ # The native-protocol Client pings before each query and reconnects by itself, so it needs no
333
+ # is_alive check, and every setting is sent per query, so there is no session state to reset.
334
+ def connect(self, details, read_only):
335
+ from clickhouse_driver import Client
336
+ client_args = {}
337
+ limit = config.query_timeout()
338
+ if limit:
339
+ # Backstop in case the server never answers; the server-side limit below is what normally fires.
340
+ client_args['send_receive_timeout'] = math.ceil(limit) + 5
341
+ return Client(connect_timeout=config.CONNECT_TIMEOUT, **client_args, **_connect_args(details))
342
+
343
+ def close(self, session):
344
+ session.conn.disconnect()
345
+
346
+ def query(self, session, sql, params, limit, offset, read_only, timeout):
347
+ from clickhouse_driver.errors import ServerException
348
+ sql, args, sliced = _prepare(sql, params, 'pyformat', limit, offset, self.DIALECT)
349
+ settings = {}
350
+ if timeout:
351
+ settings['max_execution_time'] = math.ceil(timeout) # whole seconds
352
+ if read_only:
353
+ settings['readonly'] = 1 # last: it forbids changing settings after it
354
+ try:
355
+ result = session.conn.execute(sql, args, with_column_types=True, settings=settings or None)
356
+ except ServerException as error:
357
+ if error.code == _CLICKHOUSE_TIMEOUT_EXCEEDED:
358
+ raise _timed_out(timeout) from None
359
+ raise
360
+ # Statements without a result set (DDL, INSERT) may not return a (rows, types) pair.
361
+ rows, column_types = result if isinstance(result, tuple) else ([], [])
362
+ if not sliced:
363
+ rows = rows[offset:offset + limit + 1]
364
+ return [c[0] for c in column_types or []], rows
365
+
366
+ def stream(self, session, sql, params, timeout):
367
+ """Unlike query(), uses execute_iter() - the native ClickHouse protocol is columnar and block-
368
+ based to begin with, so unlike MySQL/PostgreSQL this needs no special cursor mode, just the
369
+ streaming entry point instead of the buffered one. With with_column_types=True its first yielded
370
+ item is the [(name, type), ...] column list rather than a data row. Verified end-to-end (a real
371
+ server, 1M rows, watching the actual process RSS during the HTTP download): memory grows somewhat
372
+ early on and then plateaus, rather than the roughly linear growth with result size a fully-buffered
373
+ fetch shows - block-level internal buffering, not row-by-row, but still not proportional to how
374
+ large the full result is. Also verified that max_execution_time still cancels it mid-stream, same
375
+ as query()."""
376
+ from clickhouse_driver.errors import ServerException
377
+ sql, args = bind_parameters(sql, params or {}, 'pyformat', self.DIALECT)
378
+ settings = {'readonly': 1} # streaming is always read-only - see engine.stream_sql
379
+ if timeout:
380
+ settings['max_execution_time'] = math.ceil(timeout)
381
+ try:
382
+ iterator = session.conn.execute_iter(sql, args, with_column_types=True, settings=settings)
383
+ yield tuple(name for name, _type in next(iterator))
384
+ yield from iterator
385
+ except ServerException as error:
386
+ if error.code == _CLICKHOUSE_TIMEOUT_EXCEEDED:
387
+ raise _timed_out(timeout) from None
388
+ raise
389
+
390
+
391
+ class _DuckDB(_Driver):
392
+ """DuckDB: an embedded analytical database (own storage, own `.db` file, no server process) that can
393
+ also query flat files directly - a saved query's own SQL can call `read_csv('data.csv')` or
394
+ `read_json('data.json')` without any new connection fields, so this one connection type covers both
395
+ "a genuinely capable embedded database" and "drop a file and query it".
396
+
397
+ DuckDB refuses to open a connection to a file with a different `read_only` setting than a connection
398
+ already open on it in this process ("Can't open a connection to same database file with a different
399
+ configuration than existing connections"), which the pool's read-only/read-write connections-are-
400
+ pooled-separately design would trip over constantly. So, like H2 and the generic jdbc driver, every
401
+ connection here is opened read-write regardless of the caller's read_only flag, and the read-only
402
+ guarantee rests on validate_sql() alone - verified safe against a real DuckDB database, see
403
+ tests/test_duckdb.py.
404
+
405
+ Unlike H2/jdbc, DuckDB autocommits each statement by default (nothing is normally left open for
406
+ reset() to end) and its Python connection does support cancelling an in-progress statement
407
+ (`Connection.interrupt()`), so query timeouts are enforced here via a background timer, not merely
408
+ documented as unsupported.
409
+ """
410
+ DIALECT = 'duckdb'
411
+ PARAM_STYLE = 'qmark'
412
+
413
+ def connect(self, details, read_only):
414
+ import duckdb
415
+ return duckdb.connect(_resolve_db_file(details, 'DuckDB'))
416
+
417
+ def is_alive(self, session):
418
+ try:
419
+ session.conn.execute('SELECT 1')
420
+ return True
421
+ except Exception:
422
+ return False
423
+
424
+ def reset(self, session):
425
+ import duckdb
426
+ try:
427
+ session.conn.rollback()
428
+ except duckdb.TransactionException:
429
+ pass # autocommit already applied the last statement; there was nothing open to roll back
430
+
431
+ def query(self, session, sql, params, limit, offset, read_only, timeout):
432
+ import duckdb
433
+ conn = session.conn
434
+ timer = threading.Timer(timeout, conn.interrupt) if timeout else None
435
+ if timer:
436
+ timer.daemon = True
437
+ timer.start()
438
+ try:
439
+ result = _fetch_page(conn, sql, params, 'qmark', limit, offset, self.DIALECT)
440
+ except duckdb.InterruptException:
441
+ raise _timed_out(timeout) from None
442
+ finally:
443
+ if timer:
444
+ timer.cancel()
445
+ if not read_only:
446
+ conn.commit()
447
+ return result
448
+
449
+ def stream(self, session, sql, params, timeout):
450
+ """The timer only wraps execute(), not the fetchmany() loop after it: DuckDB's engine computes the
451
+ whole result relation during execute() (see the class docstring - its own kind of buffering, not
452
+ specific to streaming), so that is the only phase that can actually run long; fetchmany() calls
453
+ afterwards just read pages of an already-computed relation and are not a timeout candidate."""
454
+ import duckdb
455
+ conn = session.conn
456
+ timer = threading.Timer(timeout, conn.interrupt) if timeout else None
457
+ if timer:
458
+ timer.daemon = True
459
+ timer.start()
460
+ try:
461
+ sql, args = bind_parameters(sql, params or {}, 'qmark', self.DIALECT)
462
+ conn.execute(sql) if args is None else conn.execute(sql, args)
463
+ except duckdb.InterruptException:
464
+ raise _timed_out(timeout) from None
465
+ finally:
466
+ if timer:
467
+ timer.cancel()
468
+ if not conn.description:
469
+ yield ()
470
+ return
471
+ yield tuple(d[0] for d in conn.description)
472
+ while True:
473
+ batch = conn.fetchmany(_STREAM_BATCH)
474
+ if not batch:
475
+ return
476
+ yield from batch
477
+
478
+
479
+ def _attach_thread_as_daemon():
480
+ """Make the calling thread a *daemon* thread as far as the JVM is concerned.
481
+
482
+ jaydebeapi attaches every thread that uses H2 to the JVM as a non-daemon thread, and JPype's shutdown then
483
+ waits for those threads forever - so a server that has handled concurrent H2 requests would hang on exit.
484
+ Attaching them as daemons first (jaydebeapi only attaches threads that are not attached yet) avoids that.
485
+ """
486
+ import jpype
487
+ if jpype.isJVMStarted() and not jpype.java.lang.Thread.isAttached():
488
+ jpype.java.lang.Thread.attachAsDaemon()
489
+
490
+
491
+ _jvm_exit_hook_registered = False
492
+
493
+
494
+ def _register_jvm_exit_hook():
495
+ global _jvm_exit_hook_registered
496
+ if not _jvm_exit_hook_registered:
497
+ # JPype shuts the JVM down from its own atexit hook, which it registers when the JVM starts (just
498
+ # now, by whichever of H2/_JDBC connected first). Hooks run last-in first-out, so registering ours
499
+ # after that makes pooled connections close while the JVM is still alive - closing them afterwards
500
+ # would hang the interpreter at exit.
501
+ atexit.register(close_pooled_connections)
502
+ _jvm_exit_hook_registered = True
503
+
504
+
505
+ def _jvm_classpath():
506
+ """Every jar an H2 or jdbc connection might need on the JVM's classpath.
507
+
508
+ JPype starts exactly one JVM per process, and its classpath is fixed at that moment - jaydebeapi only
509
+ passes the jars of *this* connect() call, so if H2 (say) happens to start the JVM first, a jdbc
510
+ connection's own jar would never be on the classpath and it would fail with a class-not-found error the
511
+ first time it is used, even though nothing about its own configuration is wrong. Passing every
512
+ currently-configured jar on every connect() call means whichever connection is used first brings all of
513
+ them along. A jdbc connection added *after* the JVM has already started still needs the server
514
+ restarted before its jar takes effect - that part is unavoidable with one JVM per process.
515
+ """
516
+ jars = {config.h2_jar()}
517
+ for details in store.read_connections().values():
518
+ if details.get('db') == 'jdbc' and details.get('jar'):
519
+ jars.add(details['jar'])
520
+ return sorted(jars)
521
+
522
+
523
+ class _H2(_Driver):
524
+ DIALECT = 'h2'
525
+ PARAM_STYLE = 'qmark'
526
+
527
+ def connect(self, details, read_only):
528
+ import jaydebeapi
529
+ _attach_thread_as_daemon()
530
+ host = details.get('host') or 'localhost'
531
+ if details.get('port') and ':' not in host:
532
+ host = f"{host}:{details['port']}"
533
+ url = f"jdbc:h2:tcp://{host}/~/{details.get('database')}"
534
+ # Unlike the other drivers, H2's JDBC setReadOnly() is only a hint and does not block writes,
535
+ # so here the read-only guarantee rests on validate_sql() alone.
536
+ conn = jaydebeapi.connect('org.h2.Driver', url, [details.get('user'), details.get('password')],
537
+ _jvm_classpath())
538
+ _register_jvm_exit_hook()
539
+ return conn
540
+
541
+ def is_alive(self, session):
542
+ _attach_thread_as_daemon()
543
+ return bool(session.conn.jconn.isValid(2))
544
+
545
+ def reset(self, session):
546
+ _attach_thread_as_daemon()
547
+ session.conn.rollback()
548
+
549
+ def close(self, session):
550
+ _attach_thread_as_daemon()
551
+ session.conn.close()
552
+
553
+ def query(self, session, sql, params, limit, offset, read_only, timeout):
554
+ import jaydebeapi
555
+ _attach_thread_as_daemon()
556
+ conn = session.conn
557
+ cursor = conn.cursor()
558
+ if timeout != session.state.get('timeout'):
559
+ cursor.execute(f'SET QUERY_TIMEOUT {_millis(timeout) if timeout else 0}')
560
+ session.state['timeout'] = timeout
561
+ try:
562
+ result = _fetch_page(cursor, sql, params, 'qmark', limit, offset, self.DIALECT)
563
+ except jaydebeapi.Error as error:
564
+ # H2: "Statement was canceled or the session timed out" (error 57014 / 90051)
565
+ if timeout and any(word in str(error).lower() for word in ('canceled', 'timed out')):
566
+ raise _timed_out(timeout) from None
567
+ raise
568
+ if not read_only:
569
+ conn.commit()
570
+ return result
571
+
572
+ def stream(self, session, sql, params, timeout):
573
+ """Same JVM-thread-attachment requirement as every other H2 method (see _attach_thread_as_daemon).
574
+ Uses the same fetchmany()-batched cursor as the default _Driver.stream(); jaydebeapi exposes no way
575
+ to set the underlying JDBC ResultSet's fetch size, so unlike MySQL/PostgreSQL this does not avoid a
576
+ real, one-off memory cost proportional to the result size on the JVM side of the bridge (verified:
577
+ a jump right after execute(), before any row is fetched, that then plateaus rather than growing
578
+ further per row) - but it does still avoid this Python process's own memory growing without bound
579
+ as more rows are pulled, which the old fetchall()-everything-at-once approach could not."""
580
+ import jaydebeapi
581
+ _attach_thread_as_daemon()
582
+ conn = session.conn
583
+ if timeout != session.state.get('timeout'):
584
+ with conn.cursor() as setter:
585
+ setter.execute(f'SET QUERY_TIMEOUT {_millis(timeout) if timeout else 0}')
586
+ session.state['timeout'] = timeout
587
+ try:
588
+ yield from _stream_cursor(conn.cursor(), sql, params, 'qmark', self.DIALECT)
589
+ except jaydebeapi.Error as error:
590
+ if timeout and any(word in str(error).lower() for word in ('canceled', 'timed out')):
591
+ raise _timed_out(timeout) from None
592
+ raise
593
+
594
+
595
+ class _JDBC(_Driver):
596
+ """A generic JDBC connection for any database not covered by a dedicated driver above (Oracle, SQL
597
+ Server, DB2, Snowflake, ...) - reuses the JVM this project already embeds for H2. A connection needs
598
+ three fields the other drivers do not: ``jar`` (path to the vendor's JDBC driver jar), ``driver_class``
599
+ (its fully-qualified Java class name) and ``jdbc_url`` (the full JDBC URL - vendor URL formats vary too
600
+ much to build one generically the way the other drivers build theirs from host/port).
601
+
602
+ Two limits are inherent to embedding one JVM per process, not specific to this driver:
603
+
604
+ - No native query-timeout cancellation. The other drivers each have a way to cancel a running statement
605
+ (a SQL command, a session variable, a driver-level cursor call); there is no such thing that works
606
+ across arbitrary JDBC drivers without reaching into jaydebeapi's private internals (its ``Cursor``
607
+ only exposes the prepared statement *after* ``execute()`` both prepares and runs it, leaving no seam
608
+ to call ``setQueryTimeout()`` first). ``QUERYAPIGATE_QUERY_TIMEOUT`` is not enforced for this connection
609
+ type; a slow query here runs to completion regardless of the configured limit.
610
+ - Like H2, ``Connection.setReadOnly()`` is advisory in the JDBC specification, not something every
611
+ driver is required to enforce - the read-only guarantee rests on ``validate_sql()`` alone, same as H2.
612
+
613
+ Pagination is applied client-side (see ``sqltools.is_paginated``), since ``LIMIT``/``OFFSET`` is not
614
+ portable SQL either. Schema introspection (``GET /connections/<name>/schema``) is not supported for this
615
+ connection type - see ``schema.py``.
616
+ """
617
+ DIALECT = 'jdbc'
618
+ PARAM_STYLE = 'qmark'
619
+
620
+ def connect(self, details, read_only):
621
+ import jaydebeapi
622
+ _attach_thread_as_daemon()
623
+ jar, driver_class, url = details.get('jar'), details.get('driver_class'), details.get('jdbc_url')
624
+ if not (jar and driver_class and url):
625
+ raise ApiError("A 'jdbc' connection needs 'jar', 'driver_class' and 'jdbc_url'", 500)
626
+ conn = jaydebeapi.connect(driver_class, url, [details.get('user'), details.get('password')],
627
+ _jvm_classpath())
628
+ if read_only:
629
+ try:
630
+ conn.jconn.setReadOnly(True)
631
+ except Exception:
632
+ log.debug('setReadOnly() was not accepted by this JDBC driver; the guard still applies',
633
+ exc_info=True)
634
+ _register_jvm_exit_hook()
635
+ return conn
636
+
637
+ def is_alive(self, session):
638
+ _attach_thread_as_daemon()
639
+ return bool(session.conn.jconn.isValid(2))
640
+
641
+ def reset(self, session):
642
+ _attach_thread_as_daemon()
643
+ session.conn.rollback()
644
+
645
+ def close(self, session):
646
+ _attach_thread_as_daemon()
647
+ session.conn.close()
648
+
649
+ def query(self, session, sql, params, limit, offset, read_only, timeout):
650
+ _attach_thread_as_daemon()
651
+ conn = session.conn
652
+ cursor = conn.cursor()
653
+ result = _fetch_page(cursor, sql, params, 'qmark', limit, offset, self.DIALECT)
654
+ if not read_only:
655
+ conn.commit()
656
+ return result
657
+
658
+ def stream(self, session, sql, params, timeout):
659
+ """Same JVM-thread-attachment requirement as every other method here. No timeout to apply -
660
+ QUERYAPIGATE_QUERY_TIMEOUT is not enforced for jdbc at all (see the class docstring above)."""
661
+ _attach_thread_as_daemon()
662
+ yield from _stream_cursor(session.conn.cursor(), sql, params, 'qmark', self.DIALECT)
663
+
664
+
665
+ def _make_runner(driver):
666
+ def run(details, sql, params, limit, offset, read_only, timeout=None, pool=None):
667
+ if pool is None:
668
+ session = Session(driver.connect(details, read_only), driver)
669
+ try:
670
+ return driver.query(session, sql, params, limit, offset, read_only, timeout)
671
+ finally:
672
+ session.close()
673
+ with pool.checkout(driver, details, read_only) as session:
674
+ return driver.query(session, sql, params, limit, offset, read_only, timeout)
675
+
676
+ return run
677
+
678
+
679
+ def _make_stream_runner(driver):
680
+ """Like _make_runner, but for stream(): always opens the connection read-only (streaming never writes
681
+ - see engine.stream_sql) and, since the whole point is that a caller never has to hold the whole result
682
+ in memory, returns the driver's row generator directly instead of a materialised (columns, rows) pair.
683
+ Whichever of the two branches below checks the connection out (from the pool, or a fresh one) does not
684
+ release or close it until that generator is exhausted, raises, or is closed early - see the
685
+ "Streaming" section atop this module."""
686
+ def run(details, sql, params, timeout=None, pool=None):
687
+ if pool is None:
688
+ session = Session(driver.connect(details, True), driver)
689
+ try:
690
+ yield from driver.stream(session, sql, params, timeout)
691
+ finally:
692
+ session.close()
693
+ else:
694
+ with pool.checkout(driver, details, True) as session:
695
+ yield from driver.stream(session, sql, params, timeout)
696
+
697
+ return run
698
+
699
+
700
+ _run_mysql = _make_runner(_MySQL())
701
+ _run_postgres = _make_runner(_Postgres())
702
+ _run_clickhouse = _make_runner(_ClickHouse())
703
+ _run_h2 = _make_runner(_H2())
704
+ _run_jdbc = _make_runner(_JDBC())
705
+ _run_duckdb = _make_runner(_DuckDB())
706
+ _stream_mysql = _make_stream_runner(_MySQL())
707
+ _stream_postgres = _make_stream_runner(_Postgres())
708
+ _stream_clickhouse = _make_stream_runner(_ClickHouse())
709
+ _stream_h2 = _make_stream_runner(_H2())
710
+ _stream_jdbc = _make_stream_runner(_JDBC())
711
+ _stream_duckdb = _make_stream_runner(_DuckDB())
712
+
713
+
714
+ def _run_sqlite(details, sql, params, limit, offset, read_only, timeout=None, pool=None):
715
+ """SQLite is a local file, so opening it per call is cheap and it is never pooled."""
716
+ path = _resolve_db_file(details, 'SQLite')
717
+ if read_only:
718
+ conn = sqlite3.connect(f'{Path(path).as_uri()}?mode=ro', uri=True, timeout=config.CONNECT_TIMEOUT)
719
+ else:
720
+ conn = sqlite3.connect(path, timeout=config.CONNECT_TIMEOUT)
721
+ try:
722
+ expired = []
723
+ if timeout:
724
+ deadline = time.monotonic() + timeout
725
+
726
+ def check_deadline():
727
+ if time.monotonic() > deadline:
728
+ expired.append(True)
729
+ return 1 # non-zero aborts the running statement
730
+ return 0
731
+
732
+ conn.set_progress_handler(check_deadline, 10000) # called every 10 000 VM instructions
733
+ cursor = conn.cursor()
734
+ try:
735
+ result = _fetch_page(cursor, sql, params, 'qmark', limit, offset, 'sqlite')
736
+ except sqlite3.OperationalError:
737
+ if expired:
738
+ raise _timed_out(timeout) from None
739
+ raise
740
+ if not read_only:
741
+ conn.commit()
742
+ return result
743
+ finally:
744
+ conn.close()
745
+
746
+
747
+ def _stream_sqlite(details, sql, params, timeout=None, pool=None):
748
+ """Like _run_sqlite, never pooled - opened directly and closed once this generator is exhausted,
749
+ errors, or is closed early. Always opened read-only (streaming never writes)."""
750
+ path = _resolve_db_file(details, 'SQLite')
751
+ conn = sqlite3.connect(f'{Path(path).as_uri()}?mode=ro', uri=True, timeout=config.CONNECT_TIMEOUT)
752
+ try:
753
+ expired = []
754
+ if timeout:
755
+ deadline = time.monotonic() + timeout
756
+
757
+ def check_deadline():
758
+ if time.monotonic() > deadline:
759
+ expired.append(True)
760
+ return 1
761
+ return 0
762
+
763
+ conn.set_progress_handler(check_deadline, 10000)
764
+ cursor = conn.cursor()
765
+ try:
766
+ yield from _stream_cursor(cursor, sql, params, 'qmark', 'sqlite')
767
+ except sqlite3.OperationalError:
768
+ if expired:
769
+ raise _timed_out(timeout) from None
770
+ raise
771
+ finally:
772
+ conn.close()
773
+
774
+
775
+ RUNNERS = {
776
+ 'mysql': _run_mysql,
777
+ 'postgres': _run_postgres,
778
+ 'clickhouse': _run_clickhouse,
779
+ 'sqlite': _run_sqlite,
780
+ 'h2': _run_h2,
781
+ 'jdbc': _run_jdbc,
782
+ 'duckdb': _run_duckdb,
783
+ }
784
+
785
+ # Every dialect above supports streaming too - each stream_*() generator here yields the column names
786
+ # once, then every row, one at a time (see the "Streaming" section earlier in this module).
787
+ STREAM_RUNNERS = {
788
+ 'mysql': _stream_mysql,
789
+ 'postgres': _stream_postgres,
790
+ 'clickhouse': _stream_clickhouse,
791
+ 'sqlite': _stream_sqlite,
792
+ 'h2': _stream_h2,
793
+ 'jdbc': _stream_jdbc,
794
+ 'duckdb': _stream_duckdb,
795
+ }