arrowbricks 2.0.0__tar.gz → 3.0.1__tar.gz
This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
- {arrowbricks-2.0.0 → arrowbricks-3.0.1}/PKG-INFO +5 -5
- {arrowbricks-2.0.0 → arrowbricks-3.0.1}/README.md +4 -4
- {arrowbricks-2.0.0 → arrowbricks-3.0.1}/pyproject.toml +18 -1
- {arrowbricks-2.0.0 → arrowbricks-3.0.1}/rust/arrowbricks_core/Cargo.lock +1 -1
- {arrowbricks-2.0.0 → arrowbricks-3.0.1}/rust/arrowbricks_core/Cargo.toml +1 -1
- {arrowbricks-2.0.0 → arrowbricks-3.0.1}/rust/arrowbricks_core/README.md +2 -1
- {arrowbricks-2.0.0 → arrowbricks-3.0.1}/rust/arrowbricks_core/src/client.rs +279 -15
- {arrowbricks-2.0.0 → arrowbricks-3.0.1}/rust/arrowbricks_core/src/lib.rs +1 -1
- {arrowbricks-2.0.0 → arrowbricks-3.0.1}/rust/arrowbricks_core/src/pipeline.rs +375 -149
- {arrowbricks-2.0.0 → arrowbricks-3.0.1}/rust/arrowbricks_core/src/thrift.rs +44 -1
- {arrowbricks-2.0.0 → arrowbricks-3.0.1}/rust/arrowbricks_core/tests/wiremock_pipeline.rs +171 -26
- arrowbricks-3.0.1/rust/arrowbricks_core/tests/wiremock_thrift.rs +1335 -0
- arrowbricks-3.0.1/rust/arrowbricks_core/tests_py/conftest.py +31 -0
- {arrowbricks-2.0.0 → arrowbricks-3.0.1}/rust/arrowbricks_core/tests_py/test_ipc_stream.py +6 -4
- {arrowbricks-2.0.0 → arrowbricks-3.0.1}/rust/arrowbricks_core/tests_py/test_parameters.py +6 -2
- {arrowbricks-2.0.0 → arrowbricks-3.0.1}/rust/arrowbricks_core/tests_py/test_stream_ndjson_lines.py +12 -4
- {arrowbricks-2.0.0 → arrowbricks-3.0.1}/rust/arrowbricks_core/tests_py/test_streaming.py +12 -4
- arrowbricks-3.0.1/rust/arrowbricks_core/tests_py/test_thrift.py +225 -0
- {arrowbricks-2.0.0 → arrowbricks-3.0.1}/rust/arrowbricks_core/tests_py/test_token_provider.py +10 -5
- {arrowbricks-2.0.0 → arrowbricks-3.0.1}/rust/arrowbricks_core/tests_py/test_volume_files.py +6 -2
- arrowbricks-3.0.1/rust/arrowbricks_core/tests_py/thrift_mock.py +399 -0
- {arrowbricks-2.0.0 → arrowbricks-3.0.1}/src/arrowbricks/_core.pyi +1 -1
- {arrowbricks-2.0.0 → arrowbricks-3.0.1}/src/arrowbricks/client.py +12 -6
- {arrowbricks-2.0.0 → arrowbricks-3.0.1}/src/arrowbricks/cursor.py +14 -9
- {arrowbricks-2.0.0 → arrowbricks-3.0.1}/LICENSE +0 -0
- {arrowbricks-2.0.0 → arrowbricks-3.0.1}/rust/arrowbricks_core/.gitignore +0 -0
- {arrowbricks-2.0.0 → arrowbricks-3.0.1}/rust/arrowbricks_core/examples/duckdb_query.py +0 -0
- {arrowbricks-2.0.0 → arrowbricks-3.0.1}/rust/arrowbricks_core/examples/fastapi_sse.py +0 -0
- {arrowbricks-2.0.0 → arrowbricks-3.0.1}/rust/arrowbricks_core/rustfmt.toml +0 -0
- {arrowbricks-2.0.0 → arrowbricks-3.0.1}/rust/arrowbricks_core/src/heartbeat.rs +0 -0
- {arrowbricks-2.0.0 → arrowbricks-3.0.1}/rust/arrowbricks_core/src/json_convert.rs +0 -0
- {arrowbricks-2.0.0 → arrowbricks-3.0.1}/rust/arrowbricks_core/tests/wiremock_volume_files.rs +0 -0
- {arrowbricks-2.0.0 → arrowbricks-3.0.1}/src/arrowbricks/__init__.py +0 -0
- {arrowbricks-2.0.0 → arrowbricks-3.0.1}/src/arrowbricks/_streaming.py +0 -0
- {arrowbricks-2.0.0 → arrowbricks-3.0.1}/src/arrowbricks/py.typed +0 -0
|
@@ -1,6 +1,6 @@
|
|
|
1
1
|
Metadata-Version: 2.4
|
|
2
2
|
Name: arrowbricks
|
|
3
|
-
Version:
|
|
3
|
+
Version: 3.0.1
|
|
4
4
|
Requires-Dist: arro3-core>=0.8 ; extra == 'arro3'
|
|
5
5
|
Provides-Extra: arro3
|
|
6
6
|
License-File: LICENSE
|
|
@@ -17,10 +17,10 @@ Project-URL: Repository, https://github.com/bmsuisse/arrowbricks
|
|
|
17
17
|
<h1 align="center">arrowbricks</h1>
|
|
18
18
|
<p align="center">Databricks SQL to Arrow, with a Rust core.</p>
|
|
19
19
|
|
|
20
|
-
Runs SQL against a Databricks SQL warehouse
|
|
20
|
+
Runs SQL against a Databricks SQL warehouse and hands you the result as Arrow -- a `Cursor` shaped like [`databricks-sql-python`](https://github.com/databricks/databricks-sql-python)'s (`execute`, `fetchone`/`fetchmany`/`fetchall`, `fetchall_arrow`/`fetchmany_arrow`), or `stream_query_json` for streaming NDJSON. Talks to Databricks over Thrift by default (`protocol="thrift"`), or the REST Statement Execution API if you prefer (`protocol="sea"`) -- see below.
|
|
21
21
|
|
|
22
22
|
- **Rust core.** Statement submit/poll, bounded-concurrency chunk fetch, the reorder buffer, and Arrow-IPC decode all run in a [PyO3](https://pyo3.rs)/[arrow-rs](https://github.com/apache/arrow-rs) extension bundled in this same package -- 1.6x-2.5x faster than a pure-Python/asyncio client on a multi-chunk result, scaling further with chunk count and concurrency where asyncio+GIL plateaus.
|
|
23
|
-
- **
|
|
23
|
+
- **Thrift by default, SEA/REST as a fully-supported alternative** (`protocol="sea"` on `connect()`/`DatabricksClient(...)`). Thrift speaks the same HiveServer2-compatible protocol `databricks-sql-connector` uses by default -- measurably faster for *small* results (a statement's data can come back inline in the very same call that submits it, instead of SEA's separate poll-then-fetch round trip), and on par with SEA for a large, multi-chunk result too (its chunk downloads fan out concurrently across the whole result, same as SEA's own bounded-concurrency fetch, not serialized batch by batch) -- never slower than SEA on any query shape tested. Confirmed against a real production warehouse; see `AGENTS.md`'s design-invariant entry for the benchmarking history behind the switch.
|
|
24
24
|
- **Compressed cloud-fetch transport by default.** Every statement requests LZ4-compressed chunk downloads (same default as the official `databricks-sql-connector`) and decompresses them in Rust before you ever see the bytes -- less data over the wire, which matters more than local decode speed for a large result. Measured ~2x faster chunk-fetch time against a real 120-column/100k-row table. Disable per-client with `compress_results=False` (`connect()`/`DatabricksClient(...)`) if your link to the warehouse is fast enough that decompression CPU time stops paying for itself.
|
|
25
25
|
- **Zero required dependencies.** `pip install arrowbricks` and go.
|
|
26
26
|
- **Bring-your-own-auth** -- a static token or your own token-refresh callable. No cloud-SDK dependency baked in.
|
|
@@ -147,7 +147,7 @@ conn = connect(host=..., warehouse_id=..., token_provider=my_token_provider)
|
|
|
147
147
|
- `Cursor.fetchall_streamed(*, total_timeout_s=None)` / `Cursor.fetchall_arrow_streamed(*, total_timeout_s=None)` -- like `fetchall()`/`fetchall_arrow()`, but yield `HEARTBEAT` while pulling chunks instead of blocking silently, then the final rows/Table -- for a caller downloading a large result over SSE who needs heartbeats (and a timeout) through the *download*, not just the initial wait. Compose with `execute_streamed` and a shared deadline if you want one combined budget across both phases (see `examples/fastapi_sse_pivot.py`).
|
|
148
148
|
- `Cursor.description` -- DB-API-style `[(name, type_name, None, None, None, None, None), ...]` after `execute()`.
|
|
149
149
|
- `client.stream_query_json(sql, **kwargs)` (or the equivalent free function `stream_query_json(client, sql, **kwargs)`) -- yields `HEARTBEAT`, then each row as a JSON string, as soon as its chunk arrives. Timestamps come out as full ISO-8601, every column key is always present (`"col":null` for a null value, never an omitted key). JSON has no literal for NaN/Infinity/-Infinity, so those come back as `"col":null` by default -- pass `non_finite_floats="string"` to get `"col":"NaN"`/`"col":"Infinity"`/`"col":"-Infinity"` instead if you need to tell them apart from a real NULL.
|
|
150
|
-
- `DatabricksClient(host, warehouse_id, *, token=None, token_provider=None, protocol="
|
|
150
|
+
- `DatabricksClient(host, warehouse_id, *, token=None, token_provider=None, protocol="thrift", ...)` -- the lower-level client `Connection` wraps. `client.upload_volume_file(volume_path, data)`/`client.delete_volume_file(volume_path)` for the Files API. Pass `protocol="sea"` to opt into the REST Statement Execution API backend instead of the default Thrift one (see above) -- `prefer_inline` (SEA-only) has no effect under `protocol="thrift"` (silent no-op, not an error), since Thrift's own inline-result mechanism already covers that case.
|
|
151
151
|
- `write_ipc_stream(table, buf)` -- writes any Arrow-C-Data-Interface-compatible object as an uncompressed Arrow-IPC stream (see below).
|
|
152
152
|
- `ReplayableArrowChunk(data: bytes, chunk_index, declared_row_count=None)` -- wraps raw Arrow-IPC stream bytes (e.g. previously downloaded and stored) so they can be read more than once via `__arrow_c_stream__` (a schema peek, then the actual scan -- DuckDB's registration path does this), and `.to_table()` for a one-shot parse. No extra dependency needed.
|
|
153
153
|
|
|
@@ -208,7 +208,7 @@ con.sql("SELECT * FROM my_table WHERE id = 42").show()
|
|
|
208
208
|
|
|
209
209
|
## Why not `databricks-sql-connector`?
|
|
210
210
|
|
|
211
|
-
The [official driver](https://github.com/databricks/databricks-sql-python) is the right choice if you need full DB-API 2.0 compatibility
|
|
211
|
+
The [official driver](https://github.com/databricks/databricks-sql-python) is the right choice if you need full DB-API 2.0 compatibility. If you just want a query result as Arrow/JSON in your own async app, it drags in a lot for that: `pandas`, `thrift`, `openpyxl`, `pybreaker`, `pyjwt`, `oauthlib`, `lz4`, `requests`, `urllib3` as hard dependencies. arrowbricks speaks the same wire protocols (Thrift by default, or the REST Statement Execution API via `protocol="sea"`) with a hand-rolled Rust implementation instead, and zero required dependencies of its own. The `Cursor` API is deliberately shaped like the official driver's so switching between them is mostly a constructor change, but arrowbricks is async throughout (`execute`, `fetchone`, etc. are all coroutines) -- there's no sync escape hatch.
|
|
212
212
|
|
|
213
213
|
## A note on Arrow IPC compression
|
|
214
214
|
|
|
@@ -5,10 +5,10 @@
|
|
|
5
5
|
<h1 align="center">arrowbricks</h1>
|
|
6
6
|
<p align="center">Databricks SQL to Arrow, with a Rust core.</p>
|
|
7
7
|
|
|
8
|
-
Runs SQL against a Databricks SQL warehouse
|
|
8
|
+
Runs SQL against a Databricks SQL warehouse and hands you the result as Arrow -- a `Cursor` shaped like [`databricks-sql-python`](https://github.com/databricks/databricks-sql-python)'s (`execute`, `fetchone`/`fetchmany`/`fetchall`, `fetchall_arrow`/`fetchmany_arrow`), or `stream_query_json` for streaming NDJSON. Talks to Databricks over Thrift by default (`protocol="thrift"`), or the REST Statement Execution API if you prefer (`protocol="sea"`) -- see below.
|
|
9
9
|
|
|
10
10
|
- **Rust core.** Statement submit/poll, bounded-concurrency chunk fetch, the reorder buffer, and Arrow-IPC decode all run in a [PyO3](https://pyo3.rs)/[arrow-rs](https://github.com/apache/arrow-rs) extension bundled in this same package -- 1.6x-2.5x faster than a pure-Python/asyncio client on a multi-chunk result, scaling further with chunk count and concurrency where asyncio+GIL plateaus.
|
|
11
|
-
- **
|
|
11
|
+
- **Thrift by default, SEA/REST as a fully-supported alternative** (`protocol="sea"` on `connect()`/`DatabricksClient(...)`). Thrift speaks the same HiveServer2-compatible protocol `databricks-sql-connector` uses by default -- measurably faster for *small* results (a statement's data can come back inline in the very same call that submits it, instead of SEA's separate poll-then-fetch round trip), and on par with SEA for a large, multi-chunk result too (its chunk downloads fan out concurrently across the whole result, same as SEA's own bounded-concurrency fetch, not serialized batch by batch) -- never slower than SEA on any query shape tested. Confirmed against a real production warehouse; see `AGENTS.md`'s design-invariant entry for the benchmarking history behind the switch.
|
|
12
12
|
- **Compressed cloud-fetch transport by default.** Every statement requests LZ4-compressed chunk downloads (same default as the official `databricks-sql-connector`) and decompresses them in Rust before you ever see the bytes -- less data over the wire, which matters more than local decode speed for a large result. Measured ~2x faster chunk-fetch time against a real 120-column/100k-row table. Disable per-client with `compress_results=False` (`connect()`/`DatabricksClient(...)`) if your link to the warehouse is fast enough that decompression CPU time stops paying for itself.
|
|
13
13
|
- **Zero required dependencies.** `pip install arrowbricks` and go.
|
|
14
14
|
- **Bring-your-own-auth** -- a static token or your own token-refresh callable. No cloud-SDK dependency baked in.
|
|
@@ -135,7 +135,7 @@ conn = connect(host=..., warehouse_id=..., token_provider=my_token_provider)
|
|
|
135
135
|
- `Cursor.fetchall_streamed(*, total_timeout_s=None)` / `Cursor.fetchall_arrow_streamed(*, total_timeout_s=None)` -- like `fetchall()`/`fetchall_arrow()`, but yield `HEARTBEAT` while pulling chunks instead of blocking silently, then the final rows/Table -- for a caller downloading a large result over SSE who needs heartbeats (and a timeout) through the *download*, not just the initial wait. Compose with `execute_streamed` and a shared deadline if you want one combined budget across both phases (see `examples/fastapi_sse_pivot.py`).
|
|
136
136
|
- `Cursor.description` -- DB-API-style `[(name, type_name, None, None, None, None, None), ...]` after `execute()`.
|
|
137
137
|
- `client.stream_query_json(sql, **kwargs)` (or the equivalent free function `stream_query_json(client, sql, **kwargs)`) -- yields `HEARTBEAT`, then each row as a JSON string, as soon as its chunk arrives. Timestamps come out as full ISO-8601, every column key is always present (`"col":null` for a null value, never an omitted key). JSON has no literal for NaN/Infinity/-Infinity, so those come back as `"col":null` by default -- pass `non_finite_floats="string"` to get `"col":"NaN"`/`"col":"Infinity"`/`"col":"-Infinity"` instead if you need to tell them apart from a real NULL.
|
|
138
|
-
- `DatabricksClient(host, warehouse_id, *, token=None, token_provider=None, protocol="
|
|
138
|
+
- `DatabricksClient(host, warehouse_id, *, token=None, token_provider=None, protocol="thrift", ...)` -- the lower-level client `Connection` wraps. `client.upload_volume_file(volume_path, data)`/`client.delete_volume_file(volume_path)` for the Files API. Pass `protocol="sea"` to opt into the REST Statement Execution API backend instead of the default Thrift one (see above) -- `prefer_inline` (SEA-only) has no effect under `protocol="thrift"` (silent no-op, not an error), since Thrift's own inline-result mechanism already covers that case.
|
|
139
139
|
- `write_ipc_stream(table, buf)` -- writes any Arrow-C-Data-Interface-compatible object as an uncompressed Arrow-IPC stream (see below).
|
|
140
140
|
- `ReplayableArrowChunk(data: bytes, chunk_index, declared_row_count=None)` -- wraps raw Arrow-IPC stream bytes (e.g. previously downloaded and stored) so they can be read more than once via `__arrow_c_stream__` (a schema peek, then the actual scan -- DuckDB's registration path does this), and `.to_table()` for a one-shot parse. No extra dependency needed.
|
|
141
141
|
|
|
@@ -196,7 +196,7 @@ con.sql("SELECT * FROM my_table WHERE id = 42").show()
|
|
|
196
196
|
|
|
197
197
|
## Why not `databricks-sql-connector`?
|
|
198
198
|
|
|
199
|
-
The [official driver](https://github.com/databricks/databricks-sql-python) is the right choice if you need full DB-API 2.0 compatibility
|
|
199
|
+
The [official driver](https://github.com/databricks/databricks-sql-python) is the right choice if you need full DB-API 2.0 compatibility. If you just want a query result as Arrow/JSON in your own async app, it drags in a lot for that: `pandas`, `thrift`, `openpyxl`, `pybreaker`, `pyjwt`, `oauthlib`, `lz4`, `requests`, `urllib3` as hard dependencies. arrowbricks speaks the same wire protocols (Thrift by default, or the REST Statement Execution API via `protocol="sea"`) with a hand-rolled Rust implementation instead, and zero required dependencies of its own. The `Cursor` API is deliberately shaped like the official driver's so switching between them is mostly a constructor change, but arrowbricks is async throughout (`execute`, `fetchone`, etc. are all coroutines) -- there's no sync escape hatch.
|
|
200
200
|
|
|
201
201
|
## A note on Arrow IPC compression
|
|
202
202
|
|
|
@@ -1,6 +1,6 @@
|
|
|
1
1
|
[project]
|
|
2
2
|
name = "arrowbricks"
|
|
3
|
-
version = "
|
|
3
|
+
version = "3.0.1"
|
|
4
4
|
description = "Runs SQL against a Databricks SQL warehouse via the Statement Execution API and hands you the result as Arrow -- a DB-API-ish Cursor (fetchone/fetchmany/fetchall/fetchall_arrow) or NDJSON streaming. Rust/PyO3 core throughout -- zero required runtime dependencies."
|
|
5
5
|
readme = "README.md"
|
|
6
6
|
license = "MIT"
|
|
@@ -52,9 +52,26 @@ include = ["LICENSE"]
|
|
|
52
52
|
# synthetic Arrow-IPC chunk bytes so tests exercise a real Arrow IPC round
|
|
53
53
|
# trip. The package itself has no runtime dependency on them (or anything
|
|
54
54
|
# else) -- see rust/arrowbricks_core's own Arrow/NDJSON encoding.
|
|
55
|
+
#
|
|
56
|
+
# databricks-sql-connector is likewise test-only: tests/conftest.py's Thrift
|
|
57
|
+
# mock server (protocol="thrift") builds/parses real TBinaryProtocol wire
|
|
58
|
+
# bytes using this package's installed, Apache-Thrift-compiler-generated
|
|
59
|
+
# `databricks.sql.thrift_api.TCLIService` module (ttypes.py's real structs +
|
|
60
|
+
# TCLIService.py's real Processor dispatch) rather than hand-rolling a
|
|
61
|
+
# second Python Thrift codec to duplicate rust/arrowbricks_core/src/
|
|
62
|
+
# thrift.rs's own hand-rolled one -- that ttypes.py module is in fact the
|
|
63
|
+
# exact ground truth thrift.rs's own field IDs were read from (see its
|
|
64
|
+
# module doc comment), so building mock responses against it directly, byte
|
|
65
|
+
# for byte, is strictly more trustworthy than a second hand-written parser
|
|
66
|
+
# that could itself have independent bugs. Never imported by the shipped
|
|
67
|
+
# package -- rust/arrowbricks_core/src/thrift.rs's own hand-rolled
|
|
68
|
+
# TBinaryProtocol (no `thrift` crate, no databricks-sql-connector) is what
|
|
69
|
+
# actually ships; see that module's doc comment for why a real dependency
|
|
70
|
+
# was deliberately not pulled in there.
|
|
55
71
|
dev = [
|
|
56
72
|
"arro3-core>=0.8",
|
|
57
73
|
"arro3-io>=0.8",
|
|
74
|
+
"databricks-sql-connector>=4.4.0",
|
|
58
75
|
"pytest>=8.0",
|
|
59
76
|
"pytest-asyncio>=0.24",
|
|
60
77
|
"ruff>=0.6",
|
|
@@ -66,7 +66,7 @@ unlike the table object itself.
|
|
|
66
66
|
|
|
67
67
|
## API
|
|
68
68
|
|
|
69
|
-
- `Client(host, warehouse_id, *, token=None, token_provider=None, chunk_fetch_concurrency=64, http_timeout=60.0, wait_timeout="30s", warehouse_start_timeout=300.0, warehouse_confirmed_running_ttl_s=30.0, compress_results=True, protocol="
|
|
69
|
+
- `Client(host, warehouse_id, *, token=None, token_provider=None, chunk_fetch_concurrency=64, http_timeout=60.0, wait_timeout="30s", warehouse_start_timeout=300.0, warehouse_confirmed_running_ttl_s=30.0, compress_results=True, protocol="thrift")` -- exactly one of `token`/`token_provider`. `token_provider` is a callable (sync or async) returning a token string, called fresh on every request, no caching. `compress_results` requests LZ4-compressed cloud-fetch chunks (see above); set `False` to opt out. `protocol="thrift"` (the default, as of this crate's own real-workspace benchmarking -- see `AGENTS.md`'s design-invariant entry) speaks the same HiveServer2-compatible Thrift-over-HTTPS protocol `databricks-sql-connector` uses by default (`thrift.rs`, a hand-rolled `TBinaryProtocol` reader/writer -- no new Cargo dependency) -- measurably faster for small queries (its `ExecuteStatement` RPC can return a small result inline via `getDirectResults`, in the same call that submits the statement) at the cost of `prefer_inline` becoming a silent no-op (Thrift has no INLINE-disposition equivalent, and doesn't need one); never slower than SEA on any query shape tested. `protocol="sea"` instead talks to the REST Statement Execution API -- still fully supported, opt in explicitly if you have a reason to prefer it. On par with SEA for a large, multi-chunk result too -- `run_thrift_fetch_loop` (`pipeline.rs`) fans its chunk downloads out across a `chunk_fetch_concurrency`-sized worker pool spanning the *whole* result (pipelined with the sequential `FetchResults` discovery calls, not serialized behind them), the same concurrency shape as SEA's own `fetch_chunks_with_backpressure` (see `AGENTS.md`'s own entry on this).
|
|
70
70
|
- `Client.execute(statement, *, catalog=None, schema=None, parameters=None, prefer_inline=False) -> ResultSet` -- submits and starts background chunk fetching without pulling anything yet. `parameters` is Databricks' own named-parameter format (`[{"name":..., "value":..., "type":...}]`), passed straight through. `prefer_inline=True` submits with `disposition=INLINE, format=JSON_ARRAY` instead, for a caller who expects a small (well under Databricks' 25 MiB inline cap) result and wants to skip the chunk-fetch round trip -- on a result too big for INLINE, or containing a column type `json_convert.rs` doesn't map (STRUCT/ARRAY-of-STRUCT/MAP/VARIANT), it transparently falls back to a second, normal `execute()` and the query runs twice. See `AGENTS.md`'s "Design invariants" section (in the root package) for the full reasoning and real-workspace verification behind this.
|
|
71
71
|
- `ResultSet.fetchmany_arrow(n) -> Table` -- pulls/decodes only as many chunks as needed for `n` rows, buffering the rest; may return fewer than `n` once exhausted.
|
|
72
72
|
- `ResultSet.fetchall_arrow() -> Table` -- drains everything remaining.
|
|
@@ -105,6 +105,7 @@ built on exactly this):
|
|
|
105
105
|
import json
|
|
106
106
|
from arrowbricks import _core
|
|
107
107
|
|
|
108
|
+
|
|
108
109
|
async def _sse(sql: str):
|
|
109
110
|
async for item in client.stream_ndjson_lines(sql, total_timeout_s=300):
|
|
110
111
|
if item is _core.HEARTBEAT:
|
|
@@ -353,6 +353,16 @@ pub struct ChunkItem {
|
|
|
353
353
|
pub blob: Bytes,
|
|
354
354
|
pub row_count: Option<i64>,
|
|
355
355
|
pub chunk_index: i64,
|
|
356
|
+
/// Set only by the Thrift cloud-fetch path (`pipeline.rs`'s
|
|
357
|
+
/// `fetch_thrift_link`) -- a declared row-count bound this chunk's own
|
|
358
|
+
/// decoded batches must be sliced down to if they exceed it, since a
|
|
359
|
+
/// Thrift `resultLinks` file (like SEA's own cloud-fetch files) can
|
|
360
|
+
/// legitimately contain more rows than its own declared count for a
|
|
361
|
+
/// `LIMIT`-bounded query (see `pipeline.rs`'s `decode_chunk_item`).
|
|
362
|
+
/// `None` for every other producer (SEA's own chunk fetch, Thrift's
|
|
363
|
+
/// inline `arrowBatches`) -- their row counts are never overshot the
|
|
364
|
+
/// same way, so there's nothing to slice.
|
|
365
|
+
pub truncate_to: Option<i64>,
|
|
356
366
|
}
|
|
357
367
|
|
|
358
368
|
/// Unwraps one downloaded chunk file's outer LZ4 Frame compression --
|
|
@@ -430,18 +440,39 @@ where
|
|
|
430
440
|
}
|
|
431
441
|
}
|
|
432
442
|
|
|
433
|
-
/// Which wire protocol/backend `execute()` talks to Databricks with --
|
|
434
|
-
///
|
|
435
|
-
///
|
|
436
|
-
///
|
|
437
|
-
///
|
|
443
|
+
/// Which wire protocol/backend `execute()` talks to Databricks with -- a
|
|
444
|
+
/// choice on `Client`/`DatabricksClient` (`protocol: "sea" | "thrift"`),
|
|
445
|
+
/// **`Thrift` is the default as of the benchmarking work documented in
|
|
446
|
+
/// AGENTS.md's own design-invariant entry** (SEA remains fully supported,
|
|
447
|
+
/// explicit `protocol="sea"`) -- it speaks the same HiveServer2-compatible
|
|
448
|
+
/// `TCLIService` protocol `databricks-sql-connector` uses by *default*
|
|
449
|
+
/// (when its own `use_sea` isn't set) -- plain HTTPS POST,
|
|
438
450
|
/// `TBinaryProtocol`-encoded, no framing beyond HTTP itself (see `thrift.rs`).
|
|
439
451
|
/// Measurably faster than this crate's own SEA path for small queries
|
|
440
452
|
/// (closing the remaining gap `prefer_inline`/SEA-session-pooling didn't --
|
|
441
453
|
/// see those entries' own closing notes in `AGENTS.md`), primarily because
|
|
442
454
|
/// `TExecuteStatementReq`'s `getDirectResults` can return a small result's
|
|
443
455
|
/// data inline in the *same* RPC that submits the statement, where SEA
|
|
444
|
-
/// always needs at least a separate poll/fetch round trip.
|
|
456
|
+
/// always needs at least a separate poll/fetch round trip. Never slower
|
|
457
|
+
/// than SEA on any query shape tested, real or mocked.
|
|
458
|
+
///
|
|
459
|
+
/// `DbClient::new`'s own internal struct literal initializes `protocol:
|
|
460
|
+
/// Protocol::Thrift` too, matching this crate's real user-facing default one
|
|
461
|
+
/// layer up (`lib.rs`'s `PyDbClient::new` `#[pyo3(signature = ...)]` and
|
|
462
|
+
/// `client.py`'s `DatabricksClient.__init__`, which both default their own
|
|
463
|
+
/// `protocol` kwarg to `"thrift"`) -- deliberately kept as one single
|
|
464
|
+
/// default rather than two independently-set ones that happened to agree:
|
|
465
|
+
/// an earlier version of this had `DbClient::new` default to `Protocol::Sea`
|
|
466
|
+
/// while only the PyO3/Python layer defaulted to `"thrift"`, on the
|
|
467
|
+
/// reasoning that Rust-only callers (this crate's own test suite) always
|
|
468
|
+
/// call `.with_protocol` explicitly anyway -- found in review that this
|
|
469
|
+
/// left a real, if narrow, foot-gun for any *future* Rust-only caller who
|
|
470
|
+
/// constructs a `DbClient` directly and forgets to call `.with_protocol`,
|
|
471
|
+
/// silently getting SEA while believing they're on the new default. Every
|
|
472
|
+
/// SEA-testing call site in this crate's own test suite already sets
|
|
473
|
+
/// `.with_protocol(Protocol::Sea)` explicitly (see `tests/wiremock_pipeline.rs`),
|
|
474
|
+
/// so making this the same default as the public-facing one costs nothing
|
|
475
|
+
/// and removes the divergence entirely.
|
|
445
476
|
#[derive(Debug, Clone, Copy, PartialEq, Eq)]
|
|
446
477
|
pub enum Protocol {
|
|
447
478
|
Sea,
|
|
@@ -479,8 +510,25 @@ pub struct DbClient {
|
|
|
479
510
|
session_pool: SessionPool,
|
|
480
511
|
pub protocol: Protocol,
|
|
481
512
|
thrift_session_pool: ThriftSessionPool,
|
|
513
|
+
/// Global budget for concurrent cloud-fetch HTTP requests, sized to
|
|
514
|
+
/// `chunk_fetch_concurrency`. Every link download takes one permit; a
|
|
515
|
+
/// download that finds spare permits (few links in flight -- exactly
|
|
516
|
+
/// what a single-chunk result hits, leaving a lone TCP stream idle for
|
|
517
|
+
/// most of the link) also claims up to `MAX_SPLIT_PARTS - 1` extra ones
|
|
518
|
+
/// and splits itself into that many parallel HTTP Range requests. When
|
|
519
|
+
/// many links are already in flight (a large multi-chunk result), the
|
|
520
|
+
/// budget is exhausted and downloads fall back to one stream each --
|
|
521
|
+
/// exactly the shape the existing worker pool was already tuned for, so
|
|
522
|
+
/// this never regresses that case. Measured against a real warehouse:
|
|
523
|
+
/// a 10k-row/1-link query went from a 1299ms median to 672ms; a
|
|
524
|
+
/// 300k-row/19-link query was neutral (8552ms vs 8561ms baseline).
|
|
525
|
+
download_slots: tokio::sync::Semaphore,
|
|
482
526
|
}
|
|
483
527
|
|
|
528
|
+
/// Upper bound on how many parallel Range requests one cloud-fetch link is
|
|
529
|
+
/// ever split into -- see `DbClient::download_slots`.
|
|
530
|
+
pub const MAX_SPLIT_PARTS: usize = 8;
|
|
531
|
+
|
|
484
532
|
/// Session pool for the Thrift backend -- same checkout/checkin shape and
|
|
485
533
|
/// the exact same two hard constraints as `SessionPool` above (a session is
|
|
486
534
|
/// created *for* one (catalog, schema) pair and can't be redirected; Thrift's
|
|
@@ -509,11 +557,37 @@ struct ThriftSessionPool {
|
|
|
509
557
|
pub(crate) const THRIFT_POLL_INTERVAL: Duration = Duration::from_millis(200);
|
|
510
558
|
/// Request hints on `TExecuteStatementReq.getDirectResults`/`TFetchResultsReq` --
|
|
511
559
|
/// how much of the result the server should try to hand back in one RPC.
|
|
512
|
-
///
|
|
513
|
-
///
|
|
514
|
-
///
|
|
560
|
+
/// `THRIFT_DIRECT_RESULTS_MAX_BYTES` is honored essentially exactly, up to a
|
|
561
|
+
/// hard server-side ceiling of ~1 GiB per response, measured in *uncompressed*
|
|
562
|
+
/// `bytesNum` (e.g. a 500k-row result whose LZ4-compressed download is only
|
|
563
|
+
/// ~302 MiB still counts as ~669 MiB against this budget) -- **this used to
|
|
564
|
+
/// say "the server decides real batch sizes regardless (same as SEA's chunk
|
|
565
|
+
/// sizes)," which was wrong and cost real round trips**: raising this from
|
|
566
|
+
/// its original 100 MiB (self-inflicted 10x throttle) to 1 GiB dropped a
|
|
567
|
+
/// `LIMIT 2000000` query's sequential `FetchResults` discovery calls from 29
|
|
568
|
+
/// down to 2, confirmed against a real workspace; values above 1 GiB
|
|
569
|
+
/// (tested up to `i64::MAX`) measured identically to 1 GiB, so that's the
|
|
570
|
+
/// real ceiling to document, not paper over with an unbounded-looking
|
|
571
|
+
/// constant. `THRIFT_DIRECT_RESULTS_MAX_ROWS`, by contrast, **does not
|
|
572
|
+
/// govern the `resultLinks` (cloud-fetch) path at all** -- confirmed by
|
|
573
|
+
/// requesting as few as 10 rows on a 2M-row query and still getting every
|
|
574
|
+
/// link back; it only bounds the small-result *inline* `arrowBatches` path
|
|
575
|
+
/// (which the server switches to independently, at roughly 2-3 MiB of
|
|
576
|
+
/// actual Arrow bytes, regardless of either hint -- so raising
|
|
577
|
+
/// `MAX_BYTES` cannot accidentally turn a medium result into a giant inline
|
|
578
|
+
/// payload). Leave `MAX_ROWS` alone; there is nothing to tune there.
|
|
579
|
+
///
|
|
580
|
+
/// One real, bounded trade-off from raising `MAX_BYTES`: each link's own
|
|
581
|
+
/// `expiryTime` is ~900s from the response that issued it, so a much larger
|
|
582
|
+
/// batch issues more not-yet-downloaded links earlier, marginally
|
|
583
|
+
/// tightening the deadline for a very slow, caller-paced consumer
|
|
584
|
+
/// (`Cursor.fetchmany`) -- bounded by the download worker pool's own
|
|
585
|
+
/// channel capacity (`chunk_fetch_concurrency`), not unbounded, and there is
|
|
586
|
+
/// no re-resolution path for an already-expired Thrift link (the
|
|
587
|
+
/// `FETCH_NEXT` cursor has already advanced past it) if this ever bites in
|
|
588
|
+
/// practice.
|
|
515
589
|
const THRIFT_DIRECT_RESULTS_MAX_ROWS: i64 = 1_000_000;
|
|
516
|
-
const THRIFT_DIRECT_RESULTS_MAX_BYTES: i64 =
|
|
590
|
+
const THRIFT_DIRECT_RESULTS_MAX_BYTES: i64 = 1024 * 1024 * 1024;
|
|
517
591
|
|
|
518
592
|
/// A SEA session (`POST /api/2.0/sql/sessions`) pinned to one (catalog,
|
|
519
593
|
/// schema) pair, reused across statement submissions instead of the
|
|
@@ -604,19 +678,23 @@ impl DbClient {
|
|
|
604
678
|
warehouse_confirmed_running_at: Mutex::new(None),
|
|
605
679
|
compress_results: true,
|
|
606
680
|
session_pool: SessionPool::default(),
|
|
607
|
-
protocol: Protocol::
|
|
681
|
+
protocol: Protocol::Thrift,
|
|
608
682
|
thrift_session_pool: ThriftSessionPool::default(),
|
|
683
|
+
download_slots: tokio::sync::Semaphore::new(64),
|
|
609
684
|
}
|
|
610
685
|
}
|
|
611
686
|
|
|
612
687
|
pub fn with_concurrency(mut self, n: usize) -> Self {
|
|
613
688
|
self.chunk_fetch_concurrency = n.max(1);
|
|
689
|
+
self.download_slots = tokio::sync::Semaphore::new(self.chunk_fetch_concurrency);
|
|
614
690
|
self
|
|
615
691
|
}
|
|
616
692
|
|
|
617
|
-
/// Selects the wire protocol/backend -- `Protocol::Sea` (
|
|
618
|
-
///
|
|
619
|
-
/// comment
|
|
693
|
+
/// Selects the wire protocol/backend -- `Protocol::Sea` (this bare
|
|
694
|
+
/// `DbClient` constructor's own internal starting value, see
|
|
695
|
+
/// `Protocol`'s own doc comment for why that's not the same thing as
|
|
696
|
+
/// "the user-facing default") or `Protocol::Thrift` (the actual
|
|
697
|
+
/// user-facing default as of this session's benchmarking work).
|
|
620
698
|
pub fn with_protocol(mut self, protocol: Protocol) -> Self {
|
|
621
699
|
self.protocol = protocol;
|
|
622
700
|
self
|
|
@@ -751,7 +829,165 @@ impl DbClient {
|
|
|
751
829
|
.await
|
|
752
830
|
}
|
|
753
831
|
|
|
754
|
-
|
|
832
|
+
/// Downloads one cloud-fetch link, splitting it across parallel HTTP
|
|
833
|
+
/// Range requests when (and only when) the shared `download_slots`
|
|
834
|
+
/// budget has room to spare -- i.e. when few links are in flight, which
|
|
835
|
+
/// is exactly the case a single-chunk result hits and where a lone TCP
|
|
836
|
+
/// stream leaves most of the link idle. See `download_slots`' own doc
|
|
837
|
+
/// comment for the measured win and why the large-result case is safe.
|
|
838
|
+
pub(crate) async fn fetch_link_bytes_budgeted(
|
|
839
|
+
self: &Arc<Self>,
|
|
840
|
+
url: &str,
|
|
841
|
+
compressed: bool,
|
|
842
|
+
) -> Result<Bytes, ApiError> {
|
|
843
|
+
// One permit per download is mandatory; the worker pool already
|
|
844
|
+
// bounds concurrent links to `chunk_fetch_concurrency`, so this
|
|
845
|
+
// never blocks in practice -- it just makes the budget accounting
|
|
846
|
+
// exact.
|
|
847
|
+
let _base = self.download_slots.acquire().await;
|
|
848
|
+
let want = (MAX_SPLIT_PARTS - 1).min(self.download_slots.available_permits());
|
|
849
|
+
let extra = if want > 0 {
|
|
850
|
+
self.download_slots.try_acquire_many(want as u32).ok()
|
|
851
|
+
} else {
|
|
852
|
+
None
|
|
853
|
+
};
|
|
854
|
+
let parts = 1 + extra.as_ref().map(|p| p.num_permits()).unwrap_or(0);
|
|
855
|
+
if parts <= 1 {
|
|
856
|
+
return self.fetch_link_bytes(url, compressed).await;
|
|
857
|
+
}
|
|
858
|
+
self.fetch_link_bytes_split(url, compressed, 1 << 20, parts as u64)
|
|
859
|
+
.await
|
|
860
|
+
}
|
|
861
|
+
|
|
862
|
+
/// Downloads one cloud-fetch link as concurrent HTTP Range requests of
|
|
863
|
+
/// `part_size` bytes each, concatenated in order. The real object size
|
|
864
|
+
/// is learned from the first range response's `Content-Range` header --
|
|
865
|
+
/// the Thrift link's own `bytesNum` is the *uncompressed* row-set size,
|
|
866
|
+
/// not the file's size on blob storage, and using it produces HTTP 416.
|
|
867
|
+
///
|
|
868
|
+
/// A server that answers the probe with `200 OK` instead of `206
|
|
869
|
+
/// Partial Content` (Range unsupported) falls back to treating the
|
|
870
|
+
/// whole response as the complete file -- correct, since it ignored
|
|
871
|
+
/// Range and sent everything. But a server that *does* answer `206`
|
|
872
|
+
/// with a `Content-Range` header this code can't parse is a different,
|
|
873
|
+
/// unsafe case: silently treating the first `part_size` bytes as the
|
|
874
|
+
/// whole file would truncate the real result with no error. That's
|
|
875
|
+
/// treated as a hard failure instead, not a silent truncation --
|
|
876
|
+
/// confirmed unreachable against real Azure Blob Storage (always
|
|
877
|
+
/// returns a well-formed `bytes start-end/total`), but this is a
|
|
878
|
+
/// third-party response shape, not something this crate controls.
|
|
879
|
+
pub(crate) async fn fetch_link_bytes_split(
|
|
880
|
+
self: &Arc<Self>,
|
|
881
|
+
url: &str,
|
|
882
|
+
compressed: bool,
|
|
883
|
+
part_size: u64,
|
|
884
|
+
max_parts: u64,
|
|
885
|
+
) -> Result<Bytes, ApiError> {
|
|
886
|
+
// First part doubles as the size probe -- same retry_call wrapping
|
|
887
|
+
// every other download in this crate gets, so a transient failure
|
|
888
|
+
// on the probe itself doesn't skip straight to a hard error.
|
|
889
|
+
let (ranged, total, head) = retry_call(|| async {
|
|
890
|
+
let resp = self
|
|
891
|
+
.http
|
|
892
|
+
.get(url)
|
|
893
|
+
.header("Range", format!("bytes=0-{}", part_size - 1))
|
|
894
|
+
.timeout(self.http_timeout)
|
|
895
|
+
.send()
|
|
896
|
+
.await
|
|
897
|
+
.map_err(|e| ApiError::from_reqwest(e, true))?;
|
|
898
|
+
let status = resp.status();
|
|
899
|
+
if !status.is_success() {
|
|
900
|
+
let text = resp.text().await.unwrap_or_default();
|
|
901
|
+
return Err(ApiError::from_status(status, &text, true));
|
|
902
|
+
}
|
|
903
|
+
let ranged = status == reqwest::StatusCode::PARTIAL_CONTENT;
|
|
904
|
+
let total: Option<u64> = resp
|
|
905
|
+
.headers()
|
|
906
|
+
.get(reqwest::header::CONTENT_RANGE)
|
|
907
|
+
.and_then(|v| v.to_str().ok())
|
|
908
|
+
.and_then(|v| v.rsplit('/').next().and_then(|t| t.parse().ok()));
|
|
909
|
+
let head = resp.bytes().await.map_err(|e| ApiError::from_reqwest(e, true))?;
|
|
910
|
+
Ok((ranged, total, head))
|
|
911
|
+
})
|
|
912
|
+
.await?;
|
|
913
|
+
|
|
914
|
+
if ranged && total.is_none() {
|
|
915
|
+
return Err(ApiError::permanent(
|
|
916
|
+
"cloud-fetch link answered a Range request with 206 Partial Content but an \
|
|
917
|
+
unparseable Content-Range header -- refusing to silently return a truncated \
|
|
918
|
+
file"
|
|
919
|
+
.to_string(),
|
|
920
|
+
));
|
|
921
|
+
}
|
|
922
|
+
|
|
923
|
+
let mut handles = Vec::new();
|
|
924
|
+
if let Some(total) = ranged.then_some(total).flatten() {
|
|
925
|
+
// Spread everything after the probe part evenly over at most
|
|
926
|
+
// max_parts-1 further requests, so no single tail request
|
|
927
|
+
// dominates the wall clock.
|
|
928
|
+
let remaining = total.saturating_sub(part_size);
|
|
929
|
+
let n_rest = remaining.div_ceil(part_size).min(max_parts.saturating_sub(1));
|
|
930
|
+
let rest_size = if n_rest == 0 { 0 } else { remaining.div_ceil(n_rest) };
|
|
931
|
+
let mut start = part_size;
|
|
932
|
+
let mut n = 1u64;
|
|
933
|
+
while start < total && n <= n_rest {
|
|
934
|
+
let end = (start + rest_size - 1).min(total - 1);
|
|
935
|
+
let this = self.clone();
|
|
936
|
+
let url = url.to_string();
|
|
937
|
+
handles.push(tokio::spawn(async move {
|
|
938
|
+
retry_call(|| async {
|
|
939
|
+
let resp = this
|
|
940
|
+
.http
|
|
941
|
+
.get(&url)
|
|
942
|
+
.header("Range", format!("bytes={start}-{end}"))
|
|
943
|
+
.timeout(this.http_timeout)
|
|
944
|
+
.send()
|
|
945
|
+
.await
|
|
946
|
+
.map_err(|e| ApiError::from_reqwest(e, true))?;
|
|
947
|
+
let status = resp.status();
|
|
948
|
+
if !status.is_success() {
|
|
949
|
+
let text = resp.text().await.unwrap_or_default();
|
|
950
|
+
return Err(ApiError::from_status(status, &text, true));
|
|
951
|
+
}
|
|
952
|
+
resp.bytes().await.map_err(|e| ApiError::from_reqwest(e, true))
|
|
953
|
+
})
|
|
954
|
+
.await
|
|
955
|
+
}));
|
|
956
|
+
start = end + 1;
|
|
957
|
+
n += 1;
|
|
958
|
+
}
|
|
959
|
+
}
|
|
960
|
+
|
|
961
|
+
let mut out = bytes::BytesMut::with_capacity(total.unwrap_or(head.len() as u64) as usize);
|
|
962
|
+
out.extend_from_slice(&head);
|
|
963
|
+
for h in handles {
|
|
964
|
+
out.extend_from_slice(&h.await.map_err(join_error)??);
|
|
965
|
+
}
|
|
966
|
+
let bytes = out.freeze();
|
|
967
|
+
if let Some(t) = total
|
|
968
|
+
&& bytes.len() as u64 != t
|
|
969
|
+
{
|
|
970
|
+
return Err(ApiError::permanent(format!(
|
|
971
|
+
"split download assembled {} bytes, expected {t}",
|
|
972
|
+
bytes.len()
|
|
973
|
+
)));
|
|
974
|
+
}
|
|
975
|
+
if !compressed {
|
|
976
|
+
return Ok(bytes);
|
|
977
|
+
}
|
|
978
|
+
tokio::task::spawn_blocking(move || decompress_lz4_frame(&bytes))
|
|
979
|
+
.await
|
|
980
|
+
.map_err(join_error)?
|
|
981
|
+
}
|
|
982
|
+
|
|
983
|
+
/// Shared by both protocols -- plain REST against `/api/2.0/sql/warehouses/{id}`,
|
|
984
|
+
/// nothing SEA- or Thrift-specific about it. SEA's own `submit_and_poll`
|
|
985
|
+
/// has always called this; the Thrift path (`pipeline::execute_lazy_thrift`)
|
|
986
|
+
/// didn't, which meant a stopped warehouse got no proactive wake on
|
|
987
|
+
/// `protocol="thrift"` (now the default) -- statement submission would
|
|
988
|
+
/// eventually surface an error instead, with no `warehouse_start_timeout`
|
|
989
|
+
/// wait for it to come up first. Fixed by calling this from both.
|
|
990
|
+
pub(crate) async fn ensure_warehouse_running(&self) -> Result<(), ApiError> {
|
|
755
991
|
{
|
|
756
992
|
let confirmed = *self.warehouse_confirmed_running_at.lock().unwrap();
|
|
757
993
|
if let Some(at) = confirmed {
|
|
@@ -1432,11 +1668,26 @@ impl DbClient {
|
|
|
1432
1668
|
};
|
|
1433
1669
|
match fetched {
|
|
1434
1670
|
Ok(blobs) => {
|
|
1671
|
+
// `meta.row_count` is the manifest's declared count for
|
|
1672
|
+
// the whole `chunk_index`, not per-blob -- safe to use as
|
|
1673
|
+
// `decode_chunk_item`'s truncation bound (same "server can
|
|
1674
|
+
// over-deliver past its declared count" protection Thrift's
|
|
1675
|
+
// resultLinks and arrowBatches paths both already have, see
|
|
1676
|
+
// AGENTS.md) only when there's exactly one blob, where
|
|
1677
|
+
// "the whole chunk's count" and "this blob's count" are the
|
|
1678
|
+
// same number. A chunk_index resolving to more than one
|
|
1679
|
+
// blob is a real but rare/defensive-coding case (see
|
|
1680
|
+
// `ChunkMeta::pre_resolved_links`'s own doc comment) whose
|
|
1681
|
+
// true per-blob row split isn't known here -- truncating
|
|
1682
|
+
// the first blob to the *whole* chunk's count would be
|
|
1683
|
+
// wrong, so those are left untruncated rather than guessed.
|
|
1684
|
+
let truncate_to = if blobs.len() == 1 { meta.row_count } else { None };
|
|
1435
1685
|
for blob in blobs {
|
|
1436
1686
|
let item = ChunkItem {
|
|
1437
1687
|
blob,
|
|
1438
1688
|
row_count: meta.row_count,
|
|
1439
1689
|
chunk_index: meta.chunk_index,
|
|
1690
|
+
truncate_to,
|
|
1440
1691
|
};
|
|
1441
1692
|
if worker_tx.send(Ok(item)).await.is_err() {
|
|
1442
1693
|
return Ok(());
|
|
@@ -1546,6 +1797,19 @@ async fn join_first_error(handles: Vec<tokio::task::JoinHandle<Result<(), ApiErr
|
|
|
1546
1797
|
mod tests {
|
|
1547
1798
|
use super::*;
|
|
1548
1799
|
|
|
1800
|
+
/// Pins the exact value, not just "some big number" -- found in review
|
|
1801
|
+
/// that nothing caught an accidental revert (e.g. during a merge
|
|
1802
|
+
/// conflict) back toward the old, too-small 100 MiB default, which
|
|
1803
|
+
/// would silently reintroduce the round-trip regression documented on
|
|
1804
|
+
/// this constant's own doc comment (a 2M-row query needing 29
|
|
1805
|
+
/// `FetchResults` calls instead of 2). 1 GiB is the real, measured
|
|
1806
|
+
/// server-side ceiling -- see that doc comment for the numbers -- so
|
|
1807
|
+
/// this isn't an arbitrary value to protect, it's the actual limit.
|
|
1808
|
+
#[test]
|
|
1809
|
+
fn thrift_direct_results_max_bytes_is_the_measured_one_gib_ceiling() {
|
|
1810
|
+
assert_eq!(THRIFT_DIRECT_RESULTS_MAX_BYTES, 1024 * 1024 * 1024);
|
|
1811
|
+
}
|
|
1812
|
+
|
|
1549
1813
|
/// Regression test for a real bug found by testing against an actual
|
|
1550
1814
|
/// Databricks workspace (not just synthetic single-frame test data): a
|
|
1551
1815
|
/// real chunk's LZ4 compression is several frames concatenated back to
|