arrowbricks 2.0.0__tar.gz → 3.0.1__tar.gz

This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
Files changed (35) hide show
  1. {arrowbricks-2.0.0 → arrowbricks-3.0.1}/PKG-INFO +5 -5
  2. {arrowbricks-2.0.0 → arrowbricks-3.0.1}/README.md +4 -4
  3. {arrowbricks-2.0.0 → arrowbricks-3.0.1}/pyproject.toml +18 -1
  4. {arrowbricks-2.0.0 → arrowbricks-3.0.1}/rust/arrowbricks_core/Cargo.lock +1 -1
  5. {arrowbricks-2.0.0 → arrowbricks-3.0.1}/rust/arrowbricks_core/Cargo.toml +1 -1
  6. {arrowbricks-2.0.0 → arrowbricks-3.0.1}/rust/arrowbricks_core/README.md +2 -1
  7. {arrowbricks-2.0.0 → arrowbricks-3.0.1}/rust/arrowbricks_core/src/client.rs +279 -15
  8. {arrowbricks-2.0.0 → arrowbricks-3.0.1}/rust/arrowbricks_core/src/lib.rs +1 -1
  9. {arrowbricks-2.0.0 → arrowbricks-3.0.1}/rust/arrowbricks_core/src/pipeline.rs +375 -149
  10. {arrowbricks-2.0.0 → arrowbricks-3.0.1}/rust/arrowbricks_core/src/thrift.rs +44 -1
  11. {arrowbricks-2.0.0 → arrowbricks-3.0.1}/rust/arrowbricks_core/tests/wiremock_pipeline.rs +171 -26
  12. arrowbricks-3.0.1/rust/arrowbricks_core/tests/wiremock_thrift.rs +1335 -0
  13. arrowbricks-3.0.1/rust/arrowbricks_core/tests_py/conftest.py +31 -0
  14. {arrowbricks-2.0.0 → arrowbricks-3.0.1}/rust/arrowbricks_core/tests_py/test_ipc_stream.py +6 -4
  15. {arrowbricks-2.0.0 → arrowbricks-3.0.1}/rust/arrowbricks_core/tests_py/test_parameters.py +6 -2
  16. {arrowbricks-2.0.0 → arrowbricks-3.0.1}/rust/arrowbricks_core/tests_py/test_stream_ndjson_lines.py +12 -4
  17. {arrowbricks-2.0.0 → arrowbricks-3.0.1}/rust/arrowbricks_core/tests_py/test_streaming.py +12 -4
  18. arrowbricks-3.0.1/rust/arrowbricks_core/tests_py/test_thrift.py +225 -0
  19. {arrowbricks-2.0.0 → arrowbricks-3.0.1}/rust/arrowbricks_core/tests_py/test_token_provider.py +10 -5
  20. {arrowbricks-2.0.0 → arrowbricks-3.0.1}/rust/arrowbricks_core/tests_py/test_volume_files.py +6 -2
  21. arrowbricks-3.0.1/rust/arrowbricks_core/tests_py/thrift_mock.py +399 -0
  22. {arrowbricks-2.0.0 → arrowbricks-3.0.1}/src/arrowbricks/_core.pyi +1 -1
  23. {arrowbricks-2.0.0 → arrowbricks-3.0.1}/src/arrowbricks/client.py +12 -6
  24. {arrowbricks-2.0.0 → arrowbricks-3.0.1}/src/arrowbricks/cursor.py +14 -9
  25. {arrowbricks-2.0.0 → arrowbricks-3.0.1}/LICENSE +0 -0
  26. {arrowbricks-2.0.0 → arrowbricks-3.0.1}/rust/arrowbricks_core/.gitignore +0 -0
  27. {arrowbricks-2.0.0 → arrowbricks-3.0.1}/rust/arrowbricks_core/examples/duckdb_query.py +0 -0
  28. {arrowbricks-2.0.0 → arrowbricks-3.0.1}/rust/arrowbricks_core/examples/fastapi_sse.py +0 -0
  29. {arrowbricks-2.0.0 → arrowbricks-3.0.1}/rust/arrowbricks_core/rustfmt.toml +0 -0
  30. {arrowbricks-2.0.0 → arrowbricks-3.0.1}/rust/arrowbricks_core/src/heartbeat.rs +0 -0
  31. {arrowbricks-2.0.0 → arrowbricks-3.0.1}/rust/arrowbricks_core/src/json_convert.rs +0 -0
  32. {arrowbricks-2.0.0 → arrowbricks-3.0.1}/rust/arrowbricks_core/tests/wiremock_volume_files.rs +0 -0
  33. {arrowbricks-2.0.0 → arrowbricks-3.0.1}/src/arrowbricks/__init__.py +0 -0
  34. {arrowbricks-2.0.0 → arrowbricks-3.0.1}/src/arrowbricks/_streaming.py +0 -0
  35. {arrowbricks-2.0.0 → arrowbricks-3.0.1}/src/arrowbricks/py.typed +0 -0
@@ -1,6 +1,6 @@
1
1
  Metadata-Version: 2.4
2
2
  Name: arrowbricks
3
- Version: 2.0.0
3
+ Version: 3.0.1
4
4
  Requires-Dist: arro3-core>=0.8 ; extra == 'arro3'
5
5
  Provides-Extra: arro3
6
6
  License-File: LICENSE
@@ -17,10 +17,10 @@ Project-URL: Repository, https://github.com/bmsuisse/arrowbricks
17
17
  <h1 align="center">arrowbricks</h1>
18
18
  <p align="center">Databricks SQL to Arrow, with a Rust core.</p>
19
19
 
20
- Runs SQL against a Databricks SQL warehouse via the Statement Execution API and hands you the result as Arrow -- a `Cursor` shaped like [`databricks-sql-python`](https://github.com/databricks/databricks-sql-python)'s (`execute`, `fetchone`/`fetchmany`/`fetchall`, `fetchall_arrow`/`fetchmany_arrow`), or `stream_query_json` for streaming NDJSON.
20
+ Runs SQL against a Databricks SQL warehouse and hands you the result as Arrow -- a `Cursor` shaped like [`databricks-sql-python`](https://github.com/databricks/databricks-sql-python)'s (`execute`, `fetchone`/`fetchmany`/`fetchall`, `fetchall_arrow`/`fetchmany_arrow`), or `stream_query_json` for streaming NDJSON. Talks to Databricks over Thrift by default (`protocol="thrift"`), or the REST Statement Execution API if you prefer (`protocol="sea"`) -- see below.
21
21
 
22
22
  - **Rust core.** Statement submit/poll, bounded-concurrency chunk fetch, the reorder buffer, and Arrow-IPC decode all run in a [PyO3](https://pyo3.rs)/[arrow-rs](https://github.com/apache/arrow-rs) extension bundled in this same package -- 1.6x-2.5x faster than a pure-Python/asyncio client on a multi-chunk result, scaling further with chunk count and concurrency where asyncio+GIL plateaus.
23
- - **Optional Thrift protocol for small-query latency** (`protocol="thrift"` on `connect()`/`DatabricksClient(...)`, SEA/REST stays the default). Speaks the same HiveServer2-compatible protocol `databricks-sql-connector` uses by default -- measurably faster for small results, since a statement's data can come back inline in the very same call that submits it, instead of SEA's separate poll-then-fetch round trip.
23
+ - **Thrift by default, SEA/REST as a fully-supported alternative** (`protocol="sea"` on `connect()`/`DatabricksClient(...)`). Thrift speaks the same HiveServer2-compatible protocol `databricks-sql-connector` uses by default -- measurably faster for *small* results (a statement's data can come back inline in the very same call that submits it, instead of SEA's separate poll-then-fetch round trip), and on par with SEA for a large, multi-chunk result too (its chunk downloads fan out concurrently across the whole result, same as SEA's own bounded-concurrency fetch, not serialized batch by batch) -- never slower than SEA on any query shape tested. Confirmed against a real production warehouse; see `AGENTS.md`'s design-invariant entry for the benchmarking history behind the switch.
24
24
  - **Compressed cloud-fetch transport by default.** Every statement requests LZ4-compressed chunk downloads (same default as the official `databricks-sql-connector`) and decompresses them in Rust before you ever see the bytes -- less data over the wire, which matters more than local decode speed for a large result. Measured ~2x faster chunk-fetch time against a real 120-column/100k-row table. Disable per-client with `compress_results=False` (`connect()`/`DatabricksClient(...)`) if your link to the warehouse is fast enough that decompression CPU time stops paying for itself.
25
25
  - **Zero required dependencies.** `pip install arrowbricks` and go.
26
26
  - **Bring-your-own-auth** -- a static token or your own token-refresh callable. No cloud-SDK dependency baked in.
@@ -147,7 +147,7 @@ conn = connect(host=..., warehouse_id=..., token_provider=my_token_provider)
147
147
  - `Cursor.fetchall_streamed(*, total_timeout_s=None)` / `Cursor.fetchall_arrow_streamed(*, total_timeout_s=None)` -- like `fetchall()`/`fetchall_arrow()`, but yield `HEARTBEAT` while pulling chunks instead of blocking silently, then the final rows/Table -- for a caller downloading a large result over SSE who needs heartbeats (and a timeout) through the *download*, not just the initial wait. Compose with `execute_streamed` and a shared deadline if you want one combined budget across both phases (see `examples/fastapi_sse_pivot.py`).
148
148
  - `Cursor.description` -- DB-API-style `[(name, type_name, None, None, None, None, None), ...]` after `execute()`.
149
149
  - `client.stream_query_json(sql, **kwargs)` (or the equivalent free function `stream_query_json(client, sql, **kwargs)`) -- yields `HEARTBEAT`, then each row as a JSON string, as soon as its chunk arrives. Timestamps come out as full ISO-8601, every column key is always present (`"col":null` for a null value, never an omitted key). JSON has no literal for NaN/Infinity/-Infinity, so those come back as `"col":null` by default -- pass `non_finite_floats="string"` to get `"col":"NaN"`/`"col":"Infinity"`/`"col":"-Infinity"` instead if you need to tell them apart from a real NULL.
150
- - `DatabricksClient(host, warehouse_id, *, token=None, token_provider=None, protocol="sea", ...)` -- the lower-level client `Connection` wraps. `client.upload_volume_file(volume_path, data)`/`client.delete_volume_file(volume_path)` for the Files API. `protocol="thrift"` opts into the HiveServer2-compatible backend (see above) instead of the default REST/SEA one -- `prefer_inline` has no effect under it (silent no-op, not an error), since Thrift's own inline-result mechanism already covers that case.
150
+ - `DatabricksClient(host, warehouse_id, *, token=None, token_provider=None, protocol="thrift", ...)` -- the lower-level client `Connection` wraps. `client.upload_volume_file(volume_path, data)`/`client.delete_volume_file(volume_path)` for the Files API. Pass `protocol="sea"` to opt into the REST Statement Execution API backend instead of the default Thrift one (see above) -- `prefer_inline` (SEA-only) has no effect under `protocol="thrift"` (silent no-op, not an error), since Thrift's own inline-result mechanism already covers that case.
151
151
  - `write_ipc_stream(table, buf)` -- writes any Arrow-C-Data-Interface-compatible object as an uncompressed Arrow-IPC stream (see below).
152
152
  - `ReplayableArrowChunk(data: bytes, chunk_index, declared_row_count=None)` -- wraps raw Arrow-IPC stream bytes (e.g. previously downloaded and stored) so they can be read more than once via `__arrow_c_stream__` (a schema peek, then the actual scan -- DuckDB's registration path does this), and `.to_table()` for a one-shot parse. No extra dependency needed.
153
153
 
@@ -208,7 +208,7 @@ con.sql("SELECT * FROM my_table WHERE id = 42").show()
208
208
 
209
209
  ## Why not `databricks-sql-connector`?
210
210
 
211
- The [official driver](https://github.com/databricks/databricks-sql-python) is the right choice if you need full DB-API 2.0 compatibility over Databricks' Thrift/ODBC-style protocol. If you just want a query result as Arrow/JSON in your own async app, it drags in a lot for that: `pandas`, `thrift`, `openpyxl`, `pybreaker`, `pyjwt`, `oauthlib`, `lz4`, `requests`, `urllib3` as hard dependencies. arrowbricks talks to the plain REST Statement Execution API instead, with a Rust core and zero required dependencies of its own. The `Cursor` API is deliberately shaped like the official driver's so switching between them is mostly a constructor change, but arrowbricks is async throughout (`execute`, `fetchone`, etc. are all coroutines) -- there's no sync escape hatch.
211
+ The [official driver](https://github.com/databricks/databricks-sql-python) is the right choice if you need full DB-API 2.0 compatibility. If you just want a query result as Arrow/JSON in your own async app, it drags in a lot for that: `pandas`, `thrift`, `openpyxl`, `pybreaker`, `pyjwt`, `oauthlib`, `lz4`, `requests`, `urllib3` as hard dependencies. arrowbricks speaks the same wire protocols (Thrift by default, or the REST Statement Execution API via `protocol="sea"`) with a hand-rolled Rust implementation instead, and zero required dependencies of its own. The `Cursor` API is deliberately shaped like the official driver's so switching between them is mostly a constructor change, but arrowbricks is async throughout (`execute`, `fetchone`, etc. are all coroutines) -- there's no sync escape hatch.
212
212
 
213
213
  ## A note on Arrow IPC compression
214
214
 
@@ -5,10 +5,10 @@
5
5
  <h1 align="center">arrowbricks</h1>
6
6
  <p align="center">Databricks SQL to Arrow, with a Rust core.</p>
7
7
 
8
- Runs SQL against a Databricks SQL warehouse via the Statement Execution API and hands you the result as Arrow -- a `Cursor` shaped like [`databricks-sql-python`](https://github.com/databricks/databricks-sql-python)'s (`execute`, `fetchone`/`fetchmany`/`fetchall`, `fetchall_arrow`/`fetchmany_arrow`), or `stream_query_json` for streaming NDJSON.
8
+ Runs SQL against a Databricks SQL warehouse and hands you the result as Arrow -- a `Cursor` shaped like [`databricks-sql-python`](https://github.com/databricks/databricks-sql-python)'s (`execute`, `fetchone`/`fetchmany`/`fetchall`, `fetchall_arrow`/`fetchmany_arrow`), or `stream_query_json` for streaming NDJSON. Talks to Databricks over Thrift by default (`protocol="thrift"`), or the REST Statement Execution API if you prefer (`protocol="sea"`) -- see below.
9
9
 
10
10
  - **Rust core.** Statement submit/poll, bounded-concurrency chunk fetch, the reorder buffer, and Arrow-IPC decode all run in a [PyO3](https://pyo3.rs)/[arrow-rs](https://github.com/apache/arrow-rs) extension bundled in this same package -- 1.6x-2.5x faster than a pure-Python/asyncio client on a multi-chunk result, scaling further with chunk count and concurrency where asyncio+GIL plateaus.
11
- - **Optional Thrift protocol for small-query latency** (`protocol="thrift"` on `connect()`/`DatabricksClient(...)`, SEA/REST stays the default). Speaks the same HiveServer2-compatible protocol `databricks-sql-connector` uses by default -- measurably faster for small results, since a statement's data can come back inline in the very same call that submits it, instead of SEA's separate poll-then-fetch round trip.
11
+ - **Thrift by default, SEA/REST as a fully-supported alternative** (`protocol="sea"` on `connect()`/`DatabricksClient(...)`). Thrift speaks the same HiveServer2-compatible protocol `databricks-sql-connector` uses by default -- measurably faster for *small* results (a statement's data can come back inline in the very same call that submits it, instead of SEA's separate poll-then-fetch round trip), and on par with SEA for a large, multi-chunk result too (its chunk downloads fan out concurrently across the whole result, same as SEA's own bounded-concurrency fetch, not serialized batch by batch) -- never slower than SEA on any query shape tested. Confirmed against a real production warehouse; see `AGENTS.md`'s design-invariant entry for the benchmarking history behind the switch.
12
12
  - **Compressed cloud-fetch transport by default.** Every statement requests LZ4-compressed chunk downloads (same default as the official `databricks-sql-connector`) and decompresses them in Rust before you ever see the bytes -- less data over the wire, which matters more than local decode speed for a large result. Measured ~2x faster chunk-fetch time against a real 120-column/100k-row table. Disable per-client with `compress_results=False` (`connect()`/`DatabricksClient(...)`) if your link to the warehouse is fast enough that decompression CPU time stops paying for itself.
13
13
  - **Zero required dependencies.** `pip install arrowbricks` and go.
14
14
  - **Bring-your-own-auth** -- a static token or your own token-refresh callable. No cloud-SDK dependency baked in.
@@ -135,7 +135,7 @@ conn = connect(host=..., warehouse_id=..., token_provider=my_token_provider)
135
135
  - `Cursor.fetchall_streamed(*, total_timeout_s=None)` / `Cursor.fetchall_arrow_streamed(*, total_timeout_s=None)` -- like `fetchall()`/`fetchall_arrow()`, but yield `HEARTBEAT` while pulling chunks instead of blocking silently, then the final rows/Table -- for a caller downloading a large result over SSE who needs heartbeats (and a timeout) through the *download*, not just the initial wait. Compose with `execute_streamed` and a shared deadline if you want one combined budget across both phases (see `examples/fastapi_sse_pivot.py`).
136
136
  - `Cursor.description` -- DB-API-style `[(name, type_name, None, None, None, None, None), ...]` after `execute()`.
137
137
  - `client.stream_query_json(sql, **kwargs)` (or the equivalent free function `stream_query_json(client, sql, **kwargs)`) -- yields `HEARTBEAT`, then each row as a JSON string, as soon as its chunk arrives. Timestamps come out as full ISO-8601, every column key is always present (`"col":null` for a null value, never an omitted key). JSON has no literal for NaN/Infinity/-Infinity, so those come back as `"col":null` by default -- pass `non_finite_floats="string"` to get `"col":"NaN"`/`"col":"Infinity"`/`"col":"-Infinity"` instead if you need to tell them apart from a real NULL.
138
- - `DatabricksClient(host, warehouse_id, *, token=None, token_provider=None, protocol="sea", ...)` -- the lower-level client `Connection` wraps. `client.upload_volume_file(volume_path, data)`/`client.delete_volume_file(volume_path)` for the Files API. `protocol="thrift"` opts into the HiveServer2-compatible backend (see above) instead of the default REST/SEA one -- `prefer_inline` has no effect under it (silent no-op, not an error), since Thrift's own inline-result mechanism already covers that case.
138
+ - `DatabricksClient(host, warehouse_id, *, token=None, token_provider=None, protocol="thrift", ...)` -- the lower-level client `Connection` wraps. `client.upload_volume_file(volume_path, data)`/`client.delete_volume_file(volume_path)` for the Files API. Pass `protocol="sea"` to opt into the REST Statement Execution API backend instead of the default Thrift one (see above) -- `prefer_inline` (SEA-only) has no effect under `protocol="thrift"` (silent no-op, not an error), since Thrift's own inline-result mechanism already covers that case.
139
139
  - `write_ipc_stream(table, buf)` -- writes any Arrow-C-Data-Interface-compatible object as an uncompressed Arrow-IPC stream (see below).
140
140
  - `ReplayableArrowChunk(data: bytes, chunk_index, declared_row_count=None)` -- wraps raw Arrow-IPC stream bytes (e.g. previously downloaded and stored) so they can be read more than once via `__arrow_c_stream__` (a schema peek, then the actual scan -- DuckDB's registration path does this), and `.to_table()` for a one-shot parse. No extra dependency needed.
141
141
 
@@ -196,7 +196,7 @@ con.sql("SELECT * FROM my_table WHERE id = 42").show()
196
196
 
197
197
  ## Why not `databricks-sql-connector`?
198
198
 
199
- The [official driver](https://github.com/databricks/databricks-sql-python) is the right choice if you need full DB-API 2.0 compatibility over Databricks' Thrift/ODBC-style protocol. If you just want a query result as Arrow/JSON in your own async app, it drags in a lot for that: `pandas`, `thrift`, `openpyxl`, `pybreaker`, `pyjwt`, `oauthlib`, `lz4`, `requests`, `urllib3` as hard dependencies. arrowbricks talks to the plain REST Statement Execution API instead, with a Rust core and zero required dependencies of its own. The `Cursor` API is deliberately shaped like the official driver's so switching between them is mostly a constructor change, but arrowbricks is async throughout (`execute`, `fetchone`, etc. are all coroutines) -- there's no sync escape hatch.
199
+ The [official driver](https://github.com/databricks/databricks-sql-python) is the right choice if you need full DB-API 2.0 compatibility. If you just want a query result as Arrow/JSON in your own async app, it drags in a lot for that: `pandas`, `thrift`, `openpyxl`, `pybreaker`, `pyjwt`, `oauthlib`, `lz4`, `requests`, `urllib3` as hard dependencies. arrowbricks speaks the same wire protocols (Thrift by default, or the REST Statement Execution API via `protocol="sea"`) with a hand-rolled Rust implementation instead, and zero required dependencies of its own. The `Cursor` API is deliberately shaped like the official driver's so switching between them is mostly a constructor change, but arrowbricks is async throughout (`execute`, `fetchone`, etc. are all coroutines) -- there's no sync escape hatch.
200
200
 
201
201
  ## A note on Arrow IPC compression
202
202
 
@@ -1,6 +1,6 @@
1
1
  [project]
2
2
  name = "arrowbricks"
3
- version = "2.0.0"
3
+ version = "3.0.1"
4
4
  description = "Runs SQL against a Databricks SQL warehouse via the Statement Execution API and hands you the result as Arrow -- a DB-API-ish Cursor (fetchone/fetchmany/fetchall/fetchall_arrow) or NDJSON streaming. Rust/PyO3 core throughout -- zero required runtime dependencies."
5
5
  readme = "README.md"
6
6
  license = "MIT"
@@ -52,9 +52,26 @@ include = ["LICENSE"]
52
52
  # synthetic Arrow-IPC chunk bytes so tests exercise a real Arrow IPC round
53
53
  # trip. The package itself has no runtime dependency on them (or anything
54
54
  # else) -- see rust/arrowbricks_core's own Arrow/NDJSON encoding.
55
+ #
56
+ # databricks-sql-connector is likewise test-only: tests/conftest.py's Thrift
57
+ # mock server (protocol="thrift") builds/parses real TBinaryProtocol wire
58
+ # bytes using this package's installed, Apache-Thrift-compiler-generated
59
+ # `databricks.sql.thrift_api.TCLIService` module (ttypes.py's real structs +
60
+ # TCLIService.py's real Processor dispatch) rather than hand-rolling a
61
+ # second Python Thrift codec to duplicate rust/arrowbricks_core/src/
62
+ # thrift.rs's own hand-rolled one -- that ttypes.py module is in fact the
63
+ # exact ground truth thrift.rs's own field IDs were read from (see its
64
+ # module doc comment), so building mock responses against it directly, byte
65
+ # for byte, is strictly more trustworthy than a second hand-written parser
66
+ # that could itself have independent bugs. Never imported by the shipped
67
+ # package -- rust/arrowbricks_core/src/thrift.rs's own hand-rolled
68
+ # TBinaryProtocol (no `thrift` crate, no databricks-sql-connector) is what
69
+ # actually ships; see that module's doc comment for why a real dependency
70
+ # was deliberately not pulled in there.
55
71
  dev = [
56
72
  "arro3-core>=0.8",
57
73
  "arro3-io>=0.8",
74
+ "databricks-sql-connector>=4.4.0",
58
75
  "pytest>=8.0",
59
76
  "pytest-asyncio>=0.24",
60
77
  "ruff>=0.6",
@@ -240,7 +240,7 @@ dependencies = [
240
240
 
241
241
  [[package]]
242
242
  name = "arrowbricks_core"
243
- version = "2.0.0"
243
+ version = "3.0.1"
244
244
  dependencies = [
245
245
  "arrow",
246
246
  "arrow-json",
@@ -1,6 +1,6 @@
1
1
  [package]
2
2
  name = "arrowbricks_core"
3
- version = "2.0.0"
3
+ version = "3.0.1"
4
4
  edition = "2024"
5
5
  readme = "README.md"
6
6
 
@@ -66,7 +66,7 @@ unlike the table object itself.
66
66
 
67
67
  ## API
68
68
 
69
- - `Client(host, warehouse_id, *, token=None, token_provider=None, chunk_fetch_concurrency=64, http_timeout=60.0, wait_timeout="30s", warehouse_start_timeout=300.0, warehouse_confirmed_running_ttl_s=30.0, compress_results=True, protocol="sea")` -- exactly one of `token`/`token_provider`. `token_provider` is a callable (sync or async) returning a token string, called fresh on every request, no caching. `compress_results` requests LZ4-compressed cloud-fetch chunks (see above); set `False` to opt out. `protocol="sea"` (default, unchanged) talks to the REST Statement Execution API; `protocol="thrift"` instead speaks the same HiveServer2-compatible Thrift-over-HTTPS protocol `databricks-sql-connector` uses by default (`thrift.rs`, a hand-rolled `TBinaryProtocol` reader/writer -- no new Cargo dependency) -- measurably faster for small queries (its `ExecuteStatement` RPC can return a small result inline via `getDirectResults`, in the same call that submits the statement) at the cost of `prefer_inline` becoming a silent no-op (Thrift has no INLINE-disposition equivalent, and doesn't need one).
69
+ - `Client(host, warehouse_id, *, token=None, token_provider=None, chunk_fetch_concurrency=64, http_timeout=60.0, wait_timeout="30s", warehouse_start_timeout=300.0, warehouse_confirmed_running_ttl_s=30.0, compress_results=True, protocol="thrift")` -- exactly one of `token`/`token_provider`. `token_provider` is a callable (sync or async) returning a token string, called fresh on every request, no caching. `compress_results` requests LZ4-compressed cloud-fetch chunks (see above); set `False` to opt out. `protocol="thrift"` (the default, as of this crate's own real-workspace benchmarking -- see `AGENTS.md`'s design-invariant entry) speaks the same HiveServer2-compatible Thrift-over-HTTPS protocol `databricks-sql-connector` uses by default (`thrift.rs`, a hand-rolled `TBinaryProtocol` reader/writer -- no new Cargo dependency) -- measurably faster for small queries (its `ExecuteStatement` RPC can return a small result inline via `getDirectResults`, in the same call that submits the statement) at the cost of `prefer_inline` becoming a silent no-op (Thrift has no INLINE-disposition equivalent, and doesn't need one); never slower than SEA on any query shape tested. `protocol="sea"` instead talks to the REST Statement Execution API -- still fully supported, opt in explicitly if you have a reason to prefer it. On par with SEA for a large, multi-chunk result too -- `run_thrift_fetch_loop` (`pipeline.rs`) fans its chunk downloads out across a `chunk_fetch_concurrency`-sized worker pool spanning the *whole* result (pipelined with the sequential `FetchResults` discovery calls, not serialized behind them), the same concurrency shape as SEA's own `fetch_chunks_with_backpressure` (see `AGENTS.md`'s own entry on this).
70
70
  - `Client.execute(statement, *, catalog=None, schema=None, parameters=None, prefer_inline=False) -> ResultSet` -- submits and starts background chunk fetching without pulling anything yet. `parameters` is Databricks' own named-parameter format (`[{"name":..., "value":..., "type":...}]`), passed straight through. `prefer_inline=True` submits with `disposition=INLINE, format=JSON_ARRAY` instead, for a caller who expects a small (well under Databricks' 25 MiB inline cap) result and wants to skip the chunk-fetch round trip -- on a result too big for INLINE, or containing a column type `json_convert.rs` doesn't map (STRUCT/ARRAY-of-STRUCT/MAP/VARIANT), it transparently falls back to a second, normal `execute()` and the query runs twice. See `AGENTS.md`'s "Design invariants" section (in the root package) for the full reasoning and real-workspace verification behind this.
71
71
  - `ResultSet.fetchmany_arrow(n) -> Table` -- pulls/decodes only as many chunks as needed for `n` rows, buffering the rest; may return fewer than `n` once exhausted.
72
72
  - `ResultSet.fetchall_arrow() -> Table` -- drains everything remaining.
@@ -105,6 +105,7 @@ built on exactly this):
105
105
  import json
106
106
  from arrowbricks import _core
107
107
 
108
+
108
109
  async def _sse(sql: str):
109
110
  async for item in client.stream_ndjson_lines(sql, total_timeout_s=300):
110
111
  if item is _core.HEARTBEAT:
@@ -353,6 +353,16 @@ pub struct ChunkItem {
353
353
  pub blob: Bytes,
354
354
  pub row_count: Option<i64>,
355
355
  pub chunk_index: i64,
356
+ /// Set only by the Thrift cloud-fetch path (`pipeline.rs`'s
357
+ /// `fetch_thrift_link`) -- a declared row-count bound this chunk's own
358
+ /// decoded batches must be sliced down to if they exceed it, since a
359
+ /// Thrift `resultLinks` file (like SEA's own cloud-fetch files) can
360
+ /// legitimately contain more rows than its own declared count for a
361
+ /// `LIMIT`-bounded query (see `pipeline.rs`'s `decode_chunk_item`).
362
+ /// `None` for every other producer (SEA's own chunk fetch, Thrift's
363
+ /// inline `arrowBatches`) -- their row counts are never overshot the
364
+ /// same way, so there's nothing to slice.
365
+ pub truncate_to: Option<i64>,
356
366
  }
357
367
 
358
368
  /// Unwraps one downloaded chunk file's outer LZ4 Frame compression --
@@ -430,18 +440,39 @@ where
430
440
  }
431
441
  }
432
442
 
433
- /// Which wire protocol/backend `execute()` talks to Databricks with -- an
434
- /// opt-in choice on `Client`/`DatabricksClient` (`protocol: "sea" | "thrift"`),
435
- /// SEA staying the unchanged default. `Thrift` speaks the same
436
- /// HiveServer2-compatible `TCLIService` protocol `databricks-sql-connector`
437
- /// uses by *default* (when `use_sea` isn't set) -- plain HTTPS POST,
443
+ /// Which wire protocol/backend `execute()` talks to Databricks with -- a
444
+ /// choice on `Client`/`DatabricksClient` (`protocol: "sea" | "thrift"`),
445
+ /// **`Thrift` is the default as of the benchmarking work documented in
446
+ /// AGENTS.md's own design-invariant entry** (SEA remains fully supported,
447
+ /// explicit `protocol="sea"`) -- it speaks the same HiveServer2-compatible
448
+ /// `TCLIService` protocol `databricks-sql-connector` uses by *default*
449
+ /// (when its own `use_sea` isn't set) -- plain HTTPS POST,
438
450
  /// `TBinaryProtocol`-encoded, no framing beyond HTTP itself (see `thrift.rs`).
439
451
  /// Measurably faster than this crate's own SEA path for small queries
440
452
  /// (closing the remaining gap `prefer_inline`/SEA-session-pooling didn't --
441
453
  /// see those entries' own closing notes in `AGENTS.md`), primarily because
442
454
  /// `TExecuteStatementReq`'s `getDirectResults` can return a small result's
443
455
  /// data inline in the *same* RPC that submits the statement, where SEA
444
- /// always needs at least a separate poll/fetch round trip.
456
+ /// always needs at least a separate poll/fetch round trip. Never slower
457
+ /// than SEA on any query shape tested, real or mocked.
458
+ ///
459
+ /// `DbClient::new`'s own internal struct literal initializes `protocol:
460
+ /// Protocol::Thrift` too, matching this crate's real user-facing default one
461
+ /// layer up (`lib.rs`'s `PyDbClient::new` `#[pyo3(signature = ...)]` and
462
+ /// `client.py`'s `DatabricksClient.__init__`, which both default their own
463
+ /// `protocol` kwarg to `"thrift"`) -- deliberately kept as one single
464
+ /// default rather than two independently-set ones that happened to agree:
465
+ /// an earlier version of this had `DbClient::new` default to `Protocol::Sea`
466
+ /// while only the PyO3/Python layer defaulted to `"thrift"`, on the
467
+ /// reasoning that Rust-only callers (this crate's own test suite) always
468
+ /// call `.with_protocol` explicitly anyway -- found in review that this
469
+ /// left a real, if narrow, foot-gun for any *future* Rust-only caller who
470
+ /// constructs a `DbClient` directly and forgets to call `.with_protocol`,
471
+ /// silently getting SEA while believing they're on the new default. Every
472
+ /// SEA-testing call site in this crate's own test suite already sets
473
+ /// `.with_protocol(Protocol::Sea)` explicitly (see `tests/wiremock_pipeline.rs`),
474
+ /// so making this the same default as the public-facing one costs nothing
475
+ /// and removes the divergence entirely.
445
476
  #[derive(Debug, Clone, Copy, PartialEq, Eq)]
446
477
  pub enum Protocol {
447
478
  Sea,
@@ -479,8 +510,25 @@ pub struct DbClient {
479
510
  session_pool: SessionPool,
480
511
  pub protocol: Protocol,
481
512
  thrift_session_pool: ThriftSessionPool,
513
+ /// Global budget for concurrent cloud-fetch HTTP requests, sized to
514
+ /// `chunk_fetch_concurrency`. Every link download takes one permit; a
515
+ /// download that finds spare permits (few links in flight -- exactly
516
+ /// what a single-chunk result hits, leaving a lone TCP stream idle for
517
+ /// most of the link) also claims up to `MAX_SPLIT_PARTS - 1` extra ones
518
+ /// and splits itself into that many parallel HTTP Range requests. When
519
+ /// many links are already in flight (a large multi-chunk result), the
520
+ /// budget is exhausted and downloads fall back to one stream each --
521
+ /// exactly the shape the existing worker pool was already tuned for, so
522
+ /// this never regresses that case. Measured against a real warehouse:
523
+ /// a 10k-row/1-link query went from a 1299ms median to 672ms; a
524
+ /// 300k-row/19-link query was neutral (8552ms vs 8561ms baseline).
525
+ download_slots: tokio::sync::Semaphore,
482
526
  }
483
527
 
528
+ /// Upper bound on how many parallel Range requests one cloud-fetch link is
529
+ /// ever split into -- see `DbClient::download_slots`.
530
+ pub const MAX_SPLIT_PARTS: usize = 8;
531
+
484
532
  /// Session pool for the Thrift backend -- same checkout/checkin shape and
485
533
  /// the exact same two hard constraints as `SessionPool` above (a session is
486
534
  /// created *for* one (catalog, schema) pair and can't be redirected; Thrift's
@@ -509,11 +557,37 @@ struct ThriftSessionPool {
509
557
  pub(crate) const THRIFT_POLL_INTERVAL: Duration = Duration::from_millis(200);
510
558
  /// Request hints on `TExecuteStatementReq.getDirectResults`/`TFetchResultsReq` --
511
559
  /// how much of the result the server should try to hand back in one RPC.
512
- /// The server decides real batch sizes regardless (same as SEA's chunk
513
- /// sizes); these are just generous upper bounds so a small/medium result
514
- /// has a real chance of coming back in a single round trip.
560
+ /// `THRIFT_DIRECT_RESULTS_MAX_BYTES` is honored essentially exactly, up to a
561
+ /// hard server-side ceiling of ~1 GiB per response, measured in *uncompressed*
562
+ /// `bytesNum` (e.g. a 500k-row result whose LZ4-compressed download is only
563
+ /// ~302 MiB still counts as ~669 MiB against this budget) -- **this used to
564
+ /// say "the server decides real batch sizes regardless (same as SEA's chunk
565
+ /// sizes)," which was wrong and cost real round trips**: raising this from
566
+ /// its original 100 MiB (self-inflicted 10x throttle) to 1 GiB dropped a
567
+ /// `LIMIT 2000000` query's sequential `FetchResults` discovery calls from 29
568
+ /// down to 2, confirmed against a real workspace; values above 1 GiB
569
+ /// (tested up to `i64::MAX`) measured identically to 1 GiB, so that's the
570
+ /// real ceiling to document, not paper over with an unbounded-looking
571
+ /// constant. `THRIFT_DIRECT_RESULTS_MAX_ROWS`, by contrast, **does not
572
+ /// govern the `resultLinks` (cloud-fetch) path at all** -- confirmed by
573
+ /// requesting as few as 10 rows on a 2M-row query and still getting every
574
+ /// link back; it only bounds the small-result *inline* `arrowBatches` path
575
+ /// (which the server switches to independently, at roughly 2-3 MiB of
576
+ /// actual Arrow bytes, regardless of either hint -- so raising
577
+ /// `MAX_BYTES` cannot accidentally turn a medium result into a giant inline
578
+ /// payload). Leave `MAX_ROWS` alone; there is nothing to tune there.
579
+ ///
580
+ /// One real, bounded trade-off from raising `MAX_BYTES`: each link's own
581
+ /// `expiryTime` is ~900s from the response that issued it, so a much larger
582
+ /// batch issues more not-yet-downloaded links earlier, marginally
583
+ /// tightening the deadline for a very slow, caller-paced consumer
584
+ /// (`Cursor.fetchmany`) -- bounded by the download worker pool's own
585
+ /// channel capacity (`chunk_fetch_concurrency`), not unbounded, and there is
586
+ /// no re-resolution path for an already-expired Thrift link (the
587
+ /// `FETCH_NEXT` cursor has already advanced past it) if this ever bites in
588
+ /// practice.
515
589
  const THRIFT_DIRECT_RESULTS_MAX_ROWS: i64 = 1_000_000;
516
- const THRIFT_DIRECT_RESULTS_MAX_BYTES: i64 = 100 * 1024 * 1024;
590
+ const THRIFT_DIRECT_RESULTS_MAX_BYTES: i64 = 1024 * 1024 * 1024;
517
591
 
518
592
  /// A SEA session (`POST /api/2.0/sql/sessions`) pinned to one (catalog,
519
593
  /// schema) pair, reused across statement submissions instead of the
@@ -604,19 +678,23 @@ impl DbClient {
604
678
  warehouse_confirmed_running_at: Mutex::new(None),
605
679
  compress_results: true,
606
680
  session_pool: SessionPool::default(),
607
- protocol: Protocol::Sea,
681
+ protocol: Protocol::Thrift,
608
682
  thrift_session_pool: ThriftSessionPool::default(),
683
+ download_slots: tokio::sync::Semaphore::new(64),
609
684
  }
610
685
  }
611
686
 
612
687
  pub fn with_concurrency(mut self, n: usize) -> Self {
613
688
  self.chunk_fetch_concurrency = n.max(1);
689
+ self.download_slots = tokio::sync::Semaphore::new(self.chunk_fetch_concurrency);
614
690
  self
615
691
  }
616
692
 
617
- /// Selects the wire protocol/backend -- `Protocol::Sea` (default,
618
- /// unchanged) or `Protocol::Thrift` (opt-in, see `Protocol`'s own doc
619
- /// comment).
693
+ /// Selects the wire protocol/backend -- `Protocol::Sea` (this bare
694
+ /// `DbClient` constructor's own internal starting value, see
695
+ /// `Protocol`'s own doc comment for why that's not the same thing as
696
+ /// "the user-facing default") or `Protocol::Thrift` (the actual
697
+ /// user-facing default as of this session's benchmarking work).
620
698
  pub fn with_protocol(mut self, protocol: Protocol) -> Self {
621
699
  self.protocol = protocol;
622
700
  self
@@ -751,7 +829,165 @@ impl DbClient {
751
829
  .await
752
830
  }
753
831
 
754
- async fn ensure_warehouse_running(&self) -> Result<(), ApiError> {
832
+ /// Downloads one cloud-fetch link, splitting it across parallel HTTP
833
+ /// Range requests when (and only when) the shared `download_slots`
834
+ /// budget has room to spare -- i.e. when few links are in flight, which
835
+ /// is exactly the case a single-chunk result hits and where a lone TCP
836
+ /// stream leaves most of the link idle. See `download_slots`' own doc
837
+ /// comment for the measured win and why the large-result case is safe.
838
+ pub(crate) async fn fetch_link_bytes_budgeted(
839
+ self: &Arc<Self>,
840
+ url: &str,
841
+ compressed: bool,
842
+ ) -> Result<Bytes, ApiError> {
843
+ // One permit per download is mandatory; the worker pool already
844
+ // bounds concurrent links to `chunk_fetch_concurrency`, so this
845
+ // never blocks in practice -- it just makes the budget accounting
846
+ // exact.
847
+ let _base = self.download_slots.acquire().await;
848
+ let want = (MAX_SPLIT_PARTS - 1).min(self.download_slots.available_permits());
849
+ let extra = if want > 0 {
850
+ self.download_slots.try_acquire_many(want as u32).ok()
851
+ } else {
852
+ None
853
+ };
854
+ let parts = 1 + extra.as_ref().map(|p| p.num_permits()).unwrap_or(0);
855
+ if parts <= 1 {
856
+ return self.fetch_link_bytes(url, compressed).await;
857
+ }
858
+ self.fetch_link_bytes_split(url, compressed, 1 << 20, parts as u64)
859
+ .await
860
+ }
861
+
862
+ /// Downloads one cloud-fetch link as concurrent HTTP Range requests of
863
+ /// `part_size` bytes each, concatenated in order. The real object size
864
+ /// is learned from the first range response's `Content-Range` header --
865
+ /// the Thrift link's own `bytesNum` is the *uncompressed* row-set size,
866
+ /// not the file's size on blob storage, and using it produces HTTP 416.
867
+ ///
868
+ /// A server that answers the probe with `200 OK` instead of `206
869
+ /// Partial Content` (Range unsupported) falls back to treating the
870
+ /// whole response as the complete file -- correct, since it ignored
871
+ /// Range and sent everything. But a server that *does* answer `206`
872
+ /// with a `Content-Range` header this code can't parse is a different,
873
+ /// unsafe case: silently treating the first `part_size` bytes as the
874
+ /// whole file would truncate the real result with no error. That's
875
+ /// treated as a hard failure instead, not a silent truncation --
876
+ /// confirmed unreachable against real Azure Blob Storage (always
877
+ /// returns a well-formed `bytes start-end/total`), but this is a
878
+ /// third-party response shape, not something this crate controls.
879
+ pub(crate) async fn fetch_link_bytes_split(
880
+ self: &Arc<Self>,
881
+ url: &str,
882
+ compressed: bool,
883
+ part_size: u64,
884
+ max_parts: u64,
885
+ ) -> Result<Bytes, ApiError> {
886
+ // First part doubles as the size probe -- same retry_call wrapping
887
+ // every other download in this crate gets, so a transient failure
888
+ // on the probe itself doesn't skip straight to a hard error.
889
+ let (ranged, total, head) = retry_call(|| async {
890
+ let resp = self
891
+ .http
892
+ .get(url)
893
+ .header("Range", format!("bytes=0-{}", part_size - 1))
894
+ .timeout(self.http_timeout)
895
+ .send()
896
+ .await
897
+ .map_err(|e| ApiError::from_reqwest(e, true))?;
898
+ let status = resp.status();
899
+ if !status.is_success() {
900
+ let text = resp.text().await.unwrap_or_default();
901
+ return Err(ApiError::from_status(status, &text, true));
902
+ }
903
+ let ranged = status == reqwest::StatusCode::PARTIAL_CONTENT;
904
+ let total: Option<u64> = resp
905
+ .headers()
906
+ .get(reqwest::header::CONTENT_RANGE)
907
+ .and_then(|v| v.to_str().ok())
908
+ .and_then(|v| v.rsplit('/').next().and_then(|t| t.parse().ok()));
909
+ let head = resp.bytes().await.map_err(|e| ApiError::from_reqwest(e, true))?;
910
+ Ok((ranged, total, head))
911
+ })
912
+ .await?;
913
+
914
+ if ranged && total.is_none() {
915
+ return Err(ApiError::permanent(
916
+ "cloud-fetch link answered a Range request with 206 Partial Content but an \
917
+ unparseable Content-Range header -- refusing to silently return a truncated \
918
+ file"
919
+ .to_string(),
920
+ ));
921
+ }
922
+
923
+ let mut handles = Vec::new();
924
+ if let Some(total) = ranged.then_some(total).flatten() {
925
+ // Spread everything after the probe part evenly over at most
926
+ // max_parts-1 further requests, so no single tail request
927
+ // dominates the wall clock.
928
+ let remaining = total.saturating_sub(part_size);
929
+ let n_rest = remaining.div_ceil(part_size).min(max_parts.saturating_sub(1));
930
+ let rest_size = if n_rest == 0 { 0 } else { remaining.div_ceil(n_rest) };
931
+ let mut start = part_size;
932
+ let mut n = 1u64;
933
+ while start < total && n <= n_rest {
934
+ let end = (start + rest_size - 1).min(total - 1);
935
+ let this = self.clone();
936
+ let url = url.to_string();
937
+ handles.push(tokio::spawn(async move {
938
+ retry_call(|| async {
939
+ let resp = this
940
+ .http
941
+ .get(&url)
942
+ .header("Range", format!("bytes={start}-{end}"))
943
+ .timeout(this.http_timeout)
944
+ .send()
945
+ .await
946
+ .map_err(|e| ApiError::from_reqwest(e, true))?;
947
+ let status = resp.status();
948
+ if !status.is_success() {
949
+ let text = resp.text().await.unwrap_or_default();
950
+ return Err(ApiError::from_status(status, &text, true));
951
+ }
952
+ resp.bytes().await.map_err(|e| ApiError::from_reqwest(e, true))
953
+ })
954
+ .await
955
+ }));
956
+ start = end + 1;
957
+ n += 1;
958
+ }
959
+ }
960
+
961
+ let mut out = bytes::BytesMut::with_capacity(total.unwrap_or(head.len() as u64) as usize);
962
+ out.extend_from_slice(&head);
963
+ for h in handles {
964
+ out.extend_from_slice(&h.await.map_err(join_error)??);
965
+ }
966
+ let bytes = out.freeze();
967
+ if let Some(t) = total
968
+ && bytes.len() as u64 != t
969
+ {
970
+ return Err(ApiError::permanent(format!(
971
+ "split download assembled {} bytes, expected {t}",
972
+ bytes.len()
973
+ )));
974
+ }
975
+ if !compressed {
976
+ return Ok(bytes);
977
+ }
978
+ tokio::task::spawn_blocking(move || decompress_lz4_frame(&bytes))
979
+ .await
980
+ .map_err(join_error)?
981
+ }
982
+
983
+ /// Shared by both protocols -- plain REST against `/api/2.0/sql/warehouses/{id}`,
984
+ /// nothing SEA- or Thrift-specific about it. SEA's own `submit_and_poll`
985
+ /// has always called this; the Thrift path (`pipeline::execute_lazy_thrift`)
986
+ /// didn't, which meant a stopped warehouse got no proactive wake on
987
+ /// `protocol="thrift"` (now the default) -- statement submission would
988
+ /// eventually surface an error instead, with no `warehouse_start_timeout`
989
+ /// wait for it to come up first. Fixed by calling this from both.
990
+ pub(crate) async fn ensure_warehouse_running(&self) -> Result<(), ApiError> {
755
991
  {
756
992
  let confirmed = *self.warehouse_confirmed_running_at.lock().unwrap();
757
993
  if let Some(at) = confirmed {
@@ -1432,11 +1668,26 @@ impl DbClient {
1432
1668
  };
1433
1669
  match fetched {
1434
1670
  Ok(blobs) => {
1671
+ // `meta.row_count` is the manifest's declared count for
1672
+ // the whole `chunk_index`, not per-blob -- safe to use as
1673
+ // `decode_chunk_item`'s truncation bound (same "server can
1674
+ // over-deliver past its declared count" protection Thrift's
1675
+ // resultLinks and arrowBatches paths both already have, see
1676
+ // AGENTS.md) only when there's exactly one blob, where
1677
+ // "the whole chunk's count" and "this blob's count" are the
1678
+ // same number. A chunk_index resolving to more than one
1679
+ // blob is a real but rare/defensive-coding case (see
1680
+ // `ChunkMeta::pre_resolved_links`'s own doc comment) whose
1681
+ // true per-blob row split isn't known here -- truncating
1682
+ // the first blob to the *whole* chunk's count would be
1683
+ // wrong, so those are left untruncated rather than guessed.
1684
+ let truncate_to = if blobs.len() == 1 { meta.row_count } else { None };
1435
1685
  for blob in blobs {
1436
1686
  let item = ChunkItem {
1437
1687
  blob,
1438
1688
  row_count: meta.row_count,
1439
1689
  chunk_index: meta.chunk_index,
1690
+ truncate_to,
1440
1691
  };
1441
1692
  if worker_tx.send(Ok(item)).await.is_err() {
1442
1693
  return Ok(());
@@ -1546,6 +1797,19 @@ async fn join_first_error(handles: Vec<tokio::task::JoinHandle<Result<(), ApiErr
1546
1797
  mod tests {
1547
1798
  use super::*;
1548
1799
 
1800
+ /// Pins the exact value, not just "some big number" -- found in review
1801
+ /// that nothing caught an accidental revert (e.g. during a merge
1802
+ /// conflict) back toward the old, too-small 100 MiB default, which
1803
+ /// would silently reintroduce the round-trip regression documented on
1804
+ /// this constant's own doc comment (a 2M-row query needing 29
1805
+ /// `FetchResults` calls instead of 2). 1 GiB is the real, measured
1806
+ /// server-side ceiling -- see that doc comment for the numbers -- so
1807
+ /// this isn't an arbitrary value to protect, it's the actual limit.
1808
+ #[test]
1809
+ fn thrift_direct_results_max_bytes_is_the_measured_one_gib_ceiling() {
1810
+ assert_eq!(THRIFT_DIRECT_RESULTS_MAX_BYTES, 1024 * 1024 * 1024);
1811
+ }
1812
+
1549
1813
  /// Regression test for a real bug found by testing against an actual
1550
1814
  /// Databricks workspace (not just synthetic single-frame test data): a
1551
1815
  /// real chunk's LZ4 compression is several frames concatenated back to
@@ -231,7 +231,7 @@ impl PyDbClient {
231
231
  warehouse_start_timeout=300.0,
232
232
  warehouse_confirmed_running_ttl_s=30.0,
233
233
  compress_results=true,
234
- protocol="sea".to_string(),
234
+ protocol="thrift".to_string(),
235
235
  ))]
236
236
  #[allow(clippy::too_many_arguments)]
237
237
  fn new(