arrowbricks 3.1.2__tar.gz → 3.1.4__tar.gz
This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
- {arrowbricks-3.1.2 → arrowbricks-3.1.4}/PKG-INFO +40 -6
- {arrowbricks-3.1.2 → arrowbricks-3.1.4}/README.md +40 -6
- {arrowbricks-3.1.2 → arrowbricks-3.1.4}/pyproject.toml +1 -1
- {arrowbricks-3.1.2 → arrowbricks-3.1.4}/rust/arrowbricks_core/Cargo.lock +5 -64
- {arrowbricks-3.1.2 → arrowbricks-3.1.4}/rust/arrowbricks_core/Cargo.toml +20 -5
- {arrowbricks-3.1.2 → arrowbricks-3.1.4}/rust/arrowbricks_core/README.md +8 -2
- arrowbricks-3.1.4/rust/arrowbricks_core/proptest-regressions/json_convert.txt +7 -0
- {arrowbricks-3.1.2 → arrowbricks-3.1.4}/rust/arrowbricks_core/src/client/download.rs +217 -111
- {arrowbricks-3.1.2 → arrowbricks-3.1.4}/rust/arrowbricks_core/src/client.rs +2 -2
- {arrowbricks-3.1.2 → arrowbricks-3.1.4}/rust/arrowbricks_core/src/json_convert.rs +19 -8
- {arrowbricks-3.1.2 → arrowbricks-3.1.4}/rust/arrowbricks_core/src/lib.rs +12 -10
- {arrowbricks-3.1.2 → arrowbricks-3.1.4}/rust/arrowbricks_core/src/pipeline/ndjson.rs +164 -18
- {arrowbricks-3.1.2 → arrowbricks-3.1.4}/rust/arrowbricks_core/src/pipeline/reorder.rs +33 -20
- {arrowbricks-3.1.2 → arrowbricks-3.1.4}/rust/arrowbricks_core/src/pipeline/sea.rs +2 -2
- {arrowbricks-3.1.2 → arrowbricks-3.1.4}/rust/arrowbricks_core/src/pipeline/test_support.rs +3 -3
- {arrowbricks-3.1.2 → arrowbricks-3.1.4}/rust/arrowbricks_core/src/pipeline/thrift_exec.rs +33 -21
- {arrowbricks-3.1.2 → arrowbricks-3.1.4}/rust/arrowbricks_core/src/pipeline.rs +2 -0
- {arrowbricks-3.1.2 → arrowbricks-3.1.4}/rust/arrowbricks_core/src/thrift.rs +3 -3
- {arrowbricks-3.1.2 → arrowbricks-3.1.4}/rust/arrowbricks_core/tests/wiremock_pipeline.rs +4 -4
- {arrowbricks-3.1.2 → arrowbricks-3.1.4}/rust/arrowbricks_core/tests/wiremock_thrift.rs +173 -5
- arrowbricks-3.1.4/rust/arrowbricks_core/tests_py/test_ipc_stream.py +166 -0
- arrowbricks-3.1.2/rust/arrowbricks_core/tests_py/test_ipc_stream.py +0 -72
- {arrowbricks-3.1.2 → arrowbricks-3.1.4}/LICENSE +0 -0
- {arrowbricks-3.1.2 → arrowbricks-3.1.4}/rust/arrowbricks_core/.gitignore +0 -0
- {arrowbricks-3.1.2 → arrowbricks-3.1.4}/rust/arrowbricks_core/examples/duckdb_query.py +0 -0
- {arrowbricks-3.1.2 → arrowbricks-3.1.4}/rust/arrowbricks_core/examples/fastapi_sse.py +0 -0
- {arrowbricks-3.1.2 → arrowbricks-3.1.4}/rust/arrowbricks_core/rustfmt.toml +0 -0
- {arrowbricks-3.1.2 → arrowbricks-3.1.4}/rust/arrowbricks_core/src/client/error.rs +0 -0
- {arrowbricks-3.1.2 → arrowbricks-3.1.4}/rust/arrowbricks_core/src/client/model.rs +0 -0
- {arrowbricks-3.1.2 → arrowbricks-3.1.4}/rust/arrowbricks_core/src/client/sea.rs +0 -0
- {arrowbricks-3.1.2 → arrowbricks-3.1.4}/rust/arrowbricks_core/src/client/thrift_rpc.rs +0 -0
- {arrowbricks-3.1.2 → arrowbricks-3.1.4}/rust/arrowbricks_core/src/client/volume.rs +0 -0
- {arrowbricks-3.1.2 → arrowbricks-3.1.4}/rust/arrowbricks_core/src/heartbeat.rs +0 -0
- {arrowbricks-3.1.2 → arrowbricks-3.1.4}/rust/arrowbricks_core/src/pipeline/stats.rs +0 -0
- {arrowbricks-3.1.2 → arrowbricks-3.1.4}/rust/arrowbricks_core/tests/common/mod.rs +0 -0
- {arrowbricks-3.1.2 → arrowbricks-3.1.4}/rust/arrowbricks_core/tests/wiremock_volume_files.rs +0 -0
- {arrowbricks-3.1.2 → arrowbricks-3.1.4}/rust/arrowbricks_core/tests_py/conftest.py +0 -0
- {arrowbricks-3.1.2 → arrowbricks-3.1.4}/rust/arrowbricks_core/tests_py/test_parameters.py +0 -0
- {arrowbricks-3.1.2 → arrowbricks-3.1.4}/rust/arrowbricks_core/tests_py/test_stream_ndjson_lines.py +0 -0
- {arrowbricks-3.1.2 → arrowbricks-3.1.4}/rust/arrowbricks_core/tests_py/test_streaming.py +0 -0
- {arrowbricks-3.1.2 → arrowbricks-3.1.4}/rust/arrowbricks_core/tests_py/test_thrift.py +0 -0
- {arrowbricks-3.1.2 → arrowbricks-3.1.4}/rust/arrowbricks_core/tests_py/test_token_provider.py +0 -0
- {arrowbricks-3.1.2 → arrowbricks-3.1.4}/rust/arrowbricks_core/tests_py/test_volume_files.py +0 -0
- {arrowbricks-3.1.2 → arrowbricks-3.1.4}/rust/arrowbricks_core/tests_py/thrift_mock.py +0 -0
- {arrowbricks-3.1.2 → arrowbricks-3.1.4}/src/arrowbricks/__init__.py +0 -0
- {arrowbricks-3.1.2 → arrowbricks-3.1.4}/src/arrowbricks/_core.pyi +0 -0
- {arrowbricks-3.1.2 → arrowbricks-3.1.4}/src/arrowbricks/_streaming.py +0 -0
- {arrowbricks-3.1.2 → arrowbricks-3.1.4}/src/arrowbricks/client.py +0 -0
- {arrowbricks-3.1.2 → arrowbricks-3.1.4}/src/arrowbricks/cursor.py +0 -0
- {arrowbricks-3.1.2 → arrowbricks-3.1.4}/src/arrowbricks/py.typed +0 -0
|
@@ -1,6 +1,6 @@
|
|
|
1
1
|
Metadata-Version: 2.4
|
|
2
2
|
Name: arrowbricks
|
|
3
|
-
Version: 3.1.
|
|
3
|
+
Version: 3.1.4
|
|
4
4
|
Requires-Dist: arro3-core>=0.8 ; extra == 'arro3'
|
|
5
5
|
Provides-Extra: arro3
|
|
6
6
|
License-File: LICENSE
|
|
@@ -271,12 +271,17 @@ The [official driver](https://github.com/databricks/databricks-sql-python) is th
|
|
|
271
271
|
|
|
272
272
|
Measured against a real Databricks SQL warehouse (Azure Databricks, `2X-Small` **Pro** serverless warehouse, Photon on, 1-4 auto-scaling clusters -- the smallest/cheapest warehouse tier, deliberately: a bigger warehouse would narrow the gap by making the query itself slower and the client-side overhead this compares proportionally smaller). Query: `SELECT id, id * 2 AS doubled, CAST(id AS STRING) AS label FROM range(200000)` (200k rows, 3 columns), 3 timed runs after 1 discarded warm-up run, one connection reused per library:
|
|
273
273
|
|
|
274
|
-
| | avg | stdev | range |
|
|
275
|
-
|
|
276
|
-
| `databricks-sql-connector` | 0.88s | 0.06s | 0.83s - 0.95s |
|
|
277
|
-
| arrowbricks | 0.50s | 0.02s | 0.48s - 0.52s |
|
|
274
|
+
| | avg | stdev | range |
|
|
275
|
+
|---|---|---|---|
|
|
276
|
+
| `databricks-sql-connector` | 0.88s | 0.06s | 0.83s - 0.95s |
|
|
277
|
+
| arrowbricks | 0.50s | 0.02s | 0.48s - 0.52s |
|
|
278
278
|
|
|
279
|
-
|
|
279
|
+
These historical latency measurements compare the connector's `fetchall()`
|
|
280
|
+
(Python rows) with arrowbricks' `fetchall_arrow()` (Arrow), which do different
|
|
281
|
+
amounts of materialization. The previously published 16 MB memory figures
|
|
282
|
+
were invalid: the script sampled peak RSS before importing the libraries or
|
|
283
|
+
running queries. It now measures after the workload; rerun it for actual
|
|
284
|
+
peak memory on your machine. The installed-footprint measurements were:
|
|
280
285
|
|
|
281
286
|
| | installed size (package + all required deps) |
|
|
282
287
|
|---|---|
|
|
@@ -292,6 +297,35 @@ DATABRICKS_TOKEN=dapiXXXXXXXXXXXXXXXXXXXXXXXXXXXX \
|
|
|
292
297
|
python examples/benchmark_vs_connector.py
|
|
293
298
|
```
|
|
294
299
|
|
|
300
|
+
For comparing arrowbricks versions with identical APIs, use
|
|
301
|
+
[`examples/benchmark_versions.py`](examples/benchmark_versions.py). It keeps
|
|
302
|
+
one connection per version, discards warm-ups, alternates execution order,
|
|
303
|
+
passes concurrency explicitly, and checks row counts. Add `--verify-ipc`
|
|
304
|
+
(requires `arro3-core`) to compare serialized Arrow results in memory after
|
|
305
|
+
timing. Results must have stable values, schema metadata, row order, and batch boundaries;
|
|
306
|
+
verification contributes to peak process memory, and checksums are not printed. See
|
|
307
|
+
[`benchmarks/2026-09-06.md`](benchmarks/2026-09-06.md) for replay/cache measurements
|
|
308
|
+
and [`benchmarks/2026-09-06-downloads.md`](benchmarks/2026-09-06-downloads.md)
|
|
309
|
+
for the subsequent cloud-fetch scheduling measurements and limits.
|
|
310
|
+
|
|
311
|
+
Follow-up experiments cover [spare request slots](benchmarks/2026-09-06-spare-slots.md),
|
|
312
|
+
[concurrent replay and NDJSON encoding](benchmarks/2026-09-06-lowlevel.md),
|
|
313
|
+
and the rejected [LZ4 capacity](benchmarks/2026-09-06-lz4-capacity.md) and
|
|
314
|
+
[direct-fill buffer](benchmarks/2026-09-06-direct-fill.md) changes.
|
|
315
|
+
|
|
316
|
+
Cached IPC replay can be measured without a warehouse:
|
|
317
|
+
|
|
318
|
+
```bash
|
|
319
|
+
uv run python examples/benchmark_replay.py --rows 200000 --columns 16 --repeats 8
|
|
320
|
+
```
|
|
321
|
+
|
|
322
|
+
Replays now share immutable input bytes across decoded tables, reducing
|
|
323
|
+
repeated copies and memory use. Arrow may still copy misaligned fixed-width
|
|
324
|
+
buffers or decompress IPC-compressed bodies. Keeping a small slice of a
|
|
325
|
+
decoded array can retain the full source allocation until that slice is released.
|
|
326
|
+
Decoding releases the Python interpreter lock, so independent replay calls
|
|
327
|
+
can run concurrently on separate Python threads.
|
|
328
|
+
|
|
295
329
|
`BENCHMARK_SQL` overrides the query, `BENCHMARK_RUNS` (default 3) controls how many timed runs to average.
|
|
296
330
|
|
|
297
331
|
## A note on Arrow IPC compression
|
|
@@ -259,12 +259,17 @@ The [official driver](https://github.com/databricks/databricks-sql-python) is th
|
|
|
259
259
|
|
|
260
260
|
Measured against a real Databricks SQL warehouse (Azure Databricks, `2X-Small` **Pro** serverless warehouse, Photon on, 1-4 auto-scaling clusters -- the smallest/cheapest warehouse tier, deliberately: a bigger warehouse would narrow the gap by making the query itself slower and the client-side overhead this compares proportionally smaller). Query: `SELECT id, id * 2 AS doubled, CAST(id AS STRING) AS label FROM range(200000)` (200k rows, 3 columns), 3 timed runs after 1 discarded warm-up run, one connection reused per library:
|
|
261
261
|
|
|
262
|
-
| | avg | stdev | range |
|
|
263
|
-
|
|
264
|
-
| `databricks-sql-connector` | 0.88s | 0.06s | 0.83s - 0.95s |
|
|
265
|
-
| arrowbricks | 0.50s | 0.02s | 0.48s - 0.52s |
|
|
266
|
-
|
|
267
|
-
|
|
262
|
+
| | avg | stdev | range |
|
|
263
|
+
|---|---|---|---|
|
|
264
|
+
| `databricks-sql-connector` | 0.88s | 0.06s | 0.83s - 0.95s |
|
|
265
|
+
| arrowbricks | 0.50s | 0.02s | 0.48s - 0.52s |
|
|
266
|
+
|
|
267
|
+
These historical latency measurements compare the connector's `fetchall()`
|
|
268
|
+
(Python rows) with arrowbricks' `fetchall_arrow()` (Arrow), which do different
|
|
269
|
+
amounts of materialization. The previously published 16 MB memory figures
|
|
270
|
+
were invalid: the script sampled peak RSS before importing the libraries or
|
|
271
|
+
running queries. It now measures after the workload; rerun it for actual
|
|
272
|
+
peak memory on your machine. The installed-footprint measurements were:
|
|
268
273
|
|
|
269
274
|
| | installed size (package + all required deps) |
|
|
270
275
|
|---|---|
|
|
@@ -280,6 +285,35 @@ DATABRICKS_TOKEN=dapiXXXXXXXXXXXXXXXXXXXXXXXXXXXX \
|
|
|
280
285
|
python examples/benchmark_vs_connector.py
|
|
281
286
|
```
|
|
282
287
|
|
|
288
|
+
For comparing arrowbricks versions with identical APIs, use
|
|
289
|
+
[`examples/benchmark_versions.py`](examples/benchmark_versions.py). It keeps
|
|
290
|
+
one connection per version, discards warm-ups, alternates execution order,
|
|
291
|
+
passes concurrency explicitly, and checks row counts. Add `--verify-ipc`
|
|
292
|
+
(requires `arro3-core`) to compare serialized Arrow results in memory after
|
|
293
|
+
timing. Results must have stable values, schema metadata, row order, and batch boundaries;
|
|
294
|
+
verification contributes to peak process memory, and checksums are not printed. See
|
|
295
|
+
[`benchmarks/2026-09-06.md`](benchmarks/2026-09-06.md) for replay/cache measurements
|
|
296
|
+
and [`benchmarks/2026-09-06-downloads.md`](benchmarks/2026-09-06-downloads.md)
|
|
297
|
+
for the subsequent cloud-fetch scheduling measurements and limits.
|
|
298
|
+
|
|
299
|
+
Follow-up experiments cover [spare request slots](benchmarks/2026-09-06-spare-slots.md),
|
|
300
|
+
[concurrent replay and NDJSON encoding](benchmarks/2026-09-06-lowlevel.md),
|
|
301
|
+
and the rejected [LZ4 capacity](benchmarks/2026-09-06-lz4-capacity.md) and
|
|
302
|
+
[direct-fill buffer](benchmarks/2026-09-06-direct-fill.md) changes.
|
|
303
|
+
|
|
304
|
+
Cached IPC replay can be measured without a warehouse:
|
|
305
|
+
|
|
306
|
+
```bash
|
|
307
|
+
uv run python examples/benchmark_replay.py --rows 200000 --columns 16 --repeats 8
|
|
308
|
+
```
|
|
309
|
+
|
|
310
|
+
Replays now share immutable input bytes across decoded tables, reducing
|
|
311
|
+
repeated copies and memory use. Arrow may still copy misaligned fixed-width
|
|
312
|
+
buffers or decompress IPC-compressed bodies. Keeping a small slice of a
|
|
313
|
+
decoded array can retain the full source allocation until that slice is released.
|
|
314
|
+
Decoding releases the Python interpreter lock, so independent replay calls
|
|
315
|
+
can run concurrently on separate Python threads.
|
|
316
|
+
|
|
283
317
|
`BENCHMARK_SQL` overrides the query, `BENCHMARK_RUNS` (default 3) controls how many timed runs to average.
|
|
284
318
|
|
|
285
319
|
## A note on Arrow IPC compression
|
|
@@ -1,6 +1,6 @@
|
|
|
1
1
|
[project]
|
|
2
2
|
name = "arrowbricks"
|
|
3
|
-
version = "3.1.
|
|
3
|
+
version = "3.1.4"
|
|
4
4
|
description = "Runs SQL against a Databricks SQL warehouse via the Statement Execution API and hands you the result as Arrow -- a DB-API-ish Cursor (fetchone/fetchmany/fetchall/fetchall_arrow) or NDJSON streaming. Rust/PyO3 core throughout -- zero required runtime dependencies."
|
|
5
5
|
readme = "README.md"
|
|
6
6
|
license = "MIT"
|
|
@@ -34,39 +34,6 @@ dependencies = [
|
|
|
34
34
|
"libc",
|
|
35
35
|
]
|
|
36
36
|
|
|
37
|
-
[[package]]
|
|
38
|
-
name = "arrow"
|
|
39
|
-
version = "59.2.0"
|
|
40
|
-
source = "registry+https://github.com/rust-lang/crates.io-index"
|
|
41
|
-
checksum = "61d285d16bce7d0be61912f7928342b673067b6b7d7ef6cc179258ba7de1fecf"
|
|
42
|
-
dependencies = [
|
|
43
|
-
"arrow-arith",
|
|
44
|
-
"arrow-array",
|
|
45
|
-
"arrow-buffer",
|
|
46
|
-
"arrow-cast",
|
|
47
|
-
"arrow-data",
|
|
48
|
-
"arrow-ipc",
|
|
49
|
-
"arrow-ord",
|
|
50
|
-
"arrow-row",
|
|
51
|
-
"arrow-schema",
|
|
52
|
-
"arrow-select",
|
|
53
|
-
"arrow-string",
|
|
54
|
-
]
|
|
55
|
-
|
|
56
|
-
[[package]]
|
|
57
|
-
name = "arrow-arith"
|
|
58
|
-
version = "59.2.0"
|
|
59
|
-
source = "registry+https://github.com/rust-lang/crates.io-index"
|
|
60
|
-
checksum = "757ef1836251e88222542a7da2623bc1c9cb9e20afefa6db2c41e79991cd91d4"
|
|
61
|
-
dependencies = [
|
|
62
|
-
"arrow-array",
|
|
63
|
-
"arrow-buffer",
|
|
64
|
-
"arrow-data",
|
|
65
|
-
"arrow-schema",
|
|
66
|
-
"chrono",
|
|
67
|
-
"num-traits",
|
|
68
|
-
]
|
|
69
|
-
|
|
70
37
|
[[package]]
|
|
71
38
|
name = "arrow-array"
|
|
72
39
|
version = "59.2.0"
|
|
@@ -186,19 +153,6 @@ dependencies = [
|
|
|
186
153
|
"arrow-select",
|
|
187
154
|
]
|
|
188
155
|
|
|
189
|
-
[[package]]
|
|
190
|
-
name = "arrow-row"
|
|
191
|
-
version = "59.2.0"
|
|
192
|
-
source = "registry+https://github.com/rust-lang/crates.io-index"
|
|
193
|
-
checksum = "bbec439386df71ad570e6758a946111322b9e9dc8db83b5527321f0b4c9119c2"
|
|
194
|
-
dependencies = [
|
|
195
|
-
"arrow-array",
|
|
196
|
-
"arrow-buffer",
|
|
197
|
-
"arrow-data",
|
|
198
|
-
"arrow-schema",
|
|
199
|
-
"half",
|
|
200
|
-
]
|
|
201
|
-
|
|
202
156
|
[[package]]
|
|
203
157
|
name = "arrow-schema"
|
|
204
158
|
version = "59.2.0"
|
|
@@ -223,33 +177,20 @@ dependencies = [
|
|
|
223
177
|
]
|
|
224
178
|
|
|
225
179
|
[[package]]
|
|
226
|
-
name = "
|
|
227
|
-
version = "
|
|
228
|
-
source = "registry+https://github.com/rust-lang/crates.io-index"
|
|
229
|
-
checksum = "c838a25bb3691e919e0f617616ac51a4ff8517a952e29ca133cf0c22b2ce65b1"
|
|
180
|
+
name = "arrowbricks_core"
|
|
181
|
+
version = "3.1.4"
|
|
230
182
|
dependencies = [
|
|
231
183
|
"arrow-array",
|
|
232
184
|
"arrow-buffer",
|
|
233
|
-
"arrow-
|
|
234
|
-
"arrow-schema",
|
|
235
|
-
"arrow-select",
|
|
236
|
-
"memchr",
|
|
237
|
-
"num-traits",
|
|
238
|
-
"regex",
|
|
239
|
-
"regex-syntax",
|
|
240
|
-
]
|
|
241
|
-
|
|
242
|
-
[[package]]
|
|
243
|
-
name = "arrowbricks_core"
|
|
244
|
-
version = "3.1.2"
|
|
245
|
-
dependencies = [
|
|
246
|
-
"arrow",
|
|
185
|
+
"arrow-ipc",
|
|
247
186
|
"arrow-json",
|
|
187
|
+
"arrow-schema",
|
|
248
188
|
"base64 0.23.1",
|
|
249
189
|
"bytes",
|
|
250
190
|
"chrono",
|
|
251
191
|
"hyper-rustls",
|
|
252
192
|
"lz4_flex",
|
|
193
|
+
"memchr",
|
|
253
194
|
"proptest",
|
|
254
195
|
"pyo3",
|
|
255
196
|
"pyo3-arrow",
|
|
@@ -1,6 +1,6 @@
|
|
|
1
1
|
[package]
|
|
2
2
|
name = "arrowbricks_core"
|
|
3
|
-
version = "3.1.
|
|
3
|
+
version = "3.1.4"
|
|
4
4
|
edition = "2024"
|
|
5
5
|
readme = "README.md"
|
|
6
6
|
|
|
@@ -9,12 +9,16 @@ name = "arrowbricks_core"
|
|
|
9
9
|
crate-type = ["cdylib", "rlib"]
|
|
10
10
|
|
|
11
11
|
[dependencies]
|
|
12
|
-
#
|
|
13
|
-
#
|
|
14
|
-
|
|
15
|
-
arrow =
|
|
12
|
+
# Depend on the components we use. The umbrella `arrow` crate also builds
|
|
13
|
+
# arithmetic, string, and row kernels even with default-features disabled.
|
|
14
|
+
arrow-array = "59.1.0"
|
|
15
|
+
arrow-buffer = "59.1.0"
|
|
16
|
+
arrow-ipc = "59.1.0"
|
|
17
|
+
arrow-schema = "59.1.0"
|
|
16
18
|
arrow-json = "59.1.0"
|
|
17
19
|
bytes = "1.12.1"
|
|
20
|
+
# Already in the dependency graph; vectorized newline search in NDJSON output.
|
|
21
|
+
memchr = "2"
|
|
18
22
|
# Already an unconditional transitive dependency of arrow-cast (pulled in by
|
|
19
23
|
# pyo3-arrow) at this exact version -- declared directly here too so
|
|
20
24
|
# json_convert.rs can decode BINARY columns without hand-rolling base64. No
|
|
@@ -103,6 +107,11 @@ redundant_clone = "warn"
|
|
|
103
107
|
large_enum_variant = "warn"
|
|
104
108
|
needless_collect = "warn"
|
|
105
109
|
|
|
110
|
+
[profile.dev]
|
|
111
|
+
# Keep source locations in backtraces without caching full debug type data
|
|
112
|
+
# for Arrow and the HTTP stack. Incremental compilation remains enabled.
|
|
113
|
+
debug = "line-tables-only"
|
|
114
|
+
|
|
106
115
|
[profile.release]
|
|
107
116
|
# opt-level="s" over the implicit default (3) or "z" -- a size-vs-speed
|
|
108
117
|
# tradeoff previously rejected here on the reasoning that "3" was what the
|
|
@@ -126,3 +135,9 @@ opt-level = "s"
|
|
|
126
135
|
strip = true
|
|
127
136
|
lto = true
|
|
128
137
|
codegen-units = 1
|
|
138
|
+
|
|
139
|
+
[profile.release.package.pyo3-arrow]
|
|
140
|
+
# This bridge is dominated by interoperability helpers. Optimizing it for
|
|
141
|
+
# size saves ~32 KiB on macOS arm64; the IPC decoder keeps opt-level="s".
|
|
142
|
+
# Replay and live-query benchmarks are recorded in benchmarks/2026-09-06.md.
|
|
143
|
+
opt-level = "z"
|
|
@@ -32,6 +32,12 @@ This builds a mixed Python/Rust project: the compiled extension lands at
|
|
|
32
32
|
`mypy`/`pyright`/`ty` type-checking -- a compiled PyO3 extension has no type
|
|
33
33
|
info of its own without them.
|
|
34
34
|
|
|
35
|
+
Development/test builds retain source line information for backtraces but
|
|
36
|
+
omit full debug type data to reduce the Cargo cache. For full debugger
|
|
37
|
+
variable inspection, run with `CARGO_PROFILE_DEV_DEBUG=2`. Incremental
|
|
38
|
+
compilation stays enabled. Release builds optimize the Python/Arrow bridge
|
|
39
|
+
for size while keeping the IPC decoder at its existing optimization level.
|
|
40
|
+
|
|
35
41
|
## Quickstart
|
|
36
42
|
|
|
37
43
|
Importable directly (bypassing the `arrowbricks` Python wrapper) as
|
|
@@ -67,7 +73,7 @@ unlike the table object itself.
|
|
|
67
73
|
## API
|
|
68
74
|
|
|
69
75
|
- `Client(host, warehouse_id, *, token=None, token_provider=None, chunk_fetch_concurrency=64, http_timeout=60.0, wait_timeout="30s", warehouse_start_timeout=300.0, warehouse_confirmed_running_ttl_s=30.0, compress_results=True, protocol="thrift", retry_attempts=6, retry_max_wait_s=20.0)` -- exactly one of `token`/`token_provider`. `retry_attempts`/`retry_max_wait_s` tune the retry policy (total attempts, exponential-backoff ceiling in seconds) behind every retryable request this client makes; `retry_attempts` must be at least 1. `token_provider` is a callable (sync or async) returning a token string, called fresh on every request, no caching. `compress_results` requests LZ4-compressed cloud-fetch chunks (see above); set `False` to opt out. `protocol="thrift"` (the default, as of this crate's own real-workspace benchmarking -- see `AGENTS.md`'s design-invariant entry) speaks the same HiveServer2-compatible Thrift-over-HTTPS protocol `databricks-sql-connector` uses by default (`thrift.rs`, a hand-rolled `TBinaryProtocol` reader/writer -- no new Cargo dependency) -- measurably faster for small queries (its `ExecuteStatement` RPC can return a small result inline via `getDirectResults`, in the same call that submits the statement) at the cost of `prefer_inline` becoming a silent no-op (Thrift has no INLINE-disposition equivalent, and doesn't need one); never slower than SEA on any query shape tested. `protocol="sea"` instead talks to the REST Statement Execution API -- still fully supported, opt in explicitly if you have a reason to prefer it. On par with SEA for a large, multi-chunk result too -- `run_thrift_fetch_loop` (`pipeline.rs`) fans its chunk downloads out across a `chunk_fetch_concurrency`-sized worker pool spanning the *whole* result (pipelined with the sequential `FetchResults` discovery calls, not serialized behind them), the same concurrency shape as SEA's own `fetch_chunks_with_backpressure` (see `AGENTS.md`'s own entry on this).
|
|
70
|
-
- `Client.execute(statement, *, catalog=None, schema=None, parameters=None, prefer_inline=False) -> ResultSet` -- submits and starts background chunk fetching without pulling anything yet. `parameters` is Databricks' own named-parameter format (`[{"name":..., "value":..., "type":...}]`), passed straight through. `prefer_inline=True` submits with `disposition=INLINE, format=JSON_ARRAY` instead, for a caller who expects a small (well under Databricks' 25 MiB inline cap) result and wants to skip the chunk-fetch round trip --
|
|
76
|
+
- `Client.execute(statement, *, catalog=None, schema=None, parameters=None, prefer_inline=False) -> ResultSet` -- submits and starts background chunk fetching without pulling anything yet. `parameters` is Databricks' own named-parameter format (`[{"name":..., "value":..., "type":...}]`), passed straight through. `prefer_inline=True` submits with `disposition=INLINE, format=JSON_ARRAY` instead, for a caller who expects a small (well under Databricks' 25 MiB inline cap) result and wants to skip the chunk-fetch round trip -- the recognized INLINE byte-limit failure falls back to a second, normal `execute()`. If an already-succeeded result cannot be converted (including unsupported empty STRUCT arrays), it raises `ArrowbricksError` without resubmitting the statement. Use the byte-limit retry only with SQL that is safe to execute again. See `AGENTS.md`'s "Design invariants" section (in the root package) for the full reasoning and real-workspace verification behind this.
|
|
71
77
|
- `ResultSet.fetchmany_arrow(n) -> Table` -- pulls/decodes only as many chunks as needed for `n` rows, buffering the rest; may return fewer than `n` once exhausted.
|
|
72
78
|
- `ResultSet.fetchall_arrow() -> Table` -- drains everything remaining.
|
|
73
79
|
- `ResultSet.fetchall_arrow_streamed(*, total_timeout_s=None)` -- same as `fetchall_arrow()`, but an async iterator yielding the `HEARTBEAT` singleton while pulling chunks instead of blocking silently (bridge e.g. an SSE connection through the download), then a `Table` exactly once. Raises if `total_timeout_s` elapses first.
|
|
@@ -76,7 +82,7 @@ unlike the table object itself.
|
|
|
76
82
|
- `Client.stream_ndjson_lines(statement, *, catalog=None, schema=None, parameters=None, total_timeout_s=None)` -- chunk-at-a-time: an async iterator yielding the `HEARTBEAT` singleton while waiting on the statement or any individual chunk, then a `list[str]` of NDJSON lines (one per row, explicit nulls, ISO-8601 timestamps) per chunk in logical order. Decode and JSON encoding both happen in Rust -- backs `stream_query_json` end to end.
|
|
77
83
|
- `Client.upload_volume_file(volume_path, data: bytes)` / `Client.delete_volume_file(volume_path)` -- Unity Catalog volume files via the Files API. Delete treats a 404 as success (idempotent). Both raise a plain `RuntimeError` (message only) on failure.
|
|
78
84
|
- `write_ipc_stream(stream, buf)` -- free function; writes any object implementing `__arrow_c_stream__` (a `Table` from this crate, arro3, pyarrow, ...) as uncompressed Arrow-IPC stream bytes to a Python file-like object. No dependency needed regardless of the input's origin.
|
|
79
|
-
- `read_ipc_stream(data: bytes) -> Table` -- free function; the exact inverse of `write_ipc_stream`, parsing raw Arrow-IPC stream bytes back into a `Table`. No dependency needed regardless of where the bytes came from -- backs `arrowbricks.ReplayableArrowChunk`, which needs to re-parse the same cached bytes on every `__arrow_c_stream__` call.
|
|
85
|
+
- `read_ipc_stream(data: bytes) -> Table` -- free function; the exact inverse of `write_ipc_stream`, parsing raw Arrow-IPC stream bytes back into a `Table`. No dependency needed regardless of where the bytes came from -- backs `arrowbricks.ReplayableArrowChunk`, which needs to re-parse the same cached bytes on every `__arrow_c_stream__` call. Decoding releases the Python interpreter lock so independent replay calls can run concurrently; decoded arrays retain ownership of the immutable input bytes.
|
|
80
86
|
- `HEARTBEAT` -- module-level singleton; compare with `is`, e.g. `if item is _core.HEARTBEAT: ...`.
|
|
81
87
|
|
|
82
88
|
## With DuckDB
|
|
@@ -0,0 +1,7 @@
|
|
|
1
|
+
# Seeds for failure cases proptest has generated in the past. It is
|
|
2
|
+
# automatically read and these particular cases re-run before any
|
|
3
|
+
# novel cases are generated.
|
|
4
|
+
#
|
|
5
|
+
# It is recommended to check this file in to source control so that
|
|
6
|
+
# everyone who runs the test benefits from these saved cases.
|
|
7
|
+
cc c23ce2dbe22693eff6644d643de77c6c113dc918c876434391fb93a05438d706 # shrinks to type_name = "STRUCT", precision = None, scale = None, type_text = Some(""), values = []
|