arrowbricks 3.0.3__tar.gz → 3.0.4__tar.gz
This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
- {arrowbricks-3.0.3 → arrowbricks-3.0.4}/PKG-INFO +1 -1
- {arrowbricks-3.0.3 → arrowbricks-3.0.4}/pyproject.toml +1 -1
- {arrowbricks-3.0.3 → arrowbricks-3.0.4}/rust/arrowbricks_core/Cargo.lock +1 -1
- {arrowbricks-3.0.3 → arrowbricks-3.0.4}/rust/arrowbricks_core/Cargo.toml +1 -1
- {arrowbricks-3.0.3 → arrowbricks-3.0.4}/rust/arrowbricks_core/src/client.rs +35 -14
- {arrowbricks-3.0.3 → arrowbricks-3.0.4}/rust/arrowbricks_core/src/pipeline.rs +28 -3
- {arrowbricks-3.0.3 → arrowbricks-3.0.4}/src/arrowbricks/client.py +3 -1
- {arrowbricks-3.0.3 → arrowbricks-3.0.4}/LICENSE +0 -0
- {arrowbricks-3.0.3 → arrowbricks-3.0.4}/README.md +0 -0
- {arrowbricks-3.0.3 → arrowbricks-3.0.4}/rust/arrowbricks_core/.gitignore +0 -0
- {arrowbricks-3.0.3 → arrowbricks-3.0.4}/rust/arrowbricks_core/README.md +0 -0
- {arrowbricks-3.0.3 → arrowbricks-3.0.4}/rust/arrowbricks_core/examples/duckdb_query.py +0 -0
- {arrowbricks-3.0.3 → arrowbricks-3.0.4}/rust/arrowbricks_core/examples/fastapi_sse.py +0 -0
- {arrowbricks-3.0.3 → arrowbricks-3.0.4}/rust/arrowbricks_core/rustfmt.toml +0 -0
- {arrowbricks-3.0.3 → arrowbricks-3.0.4}/rust/arrowbricks_core/src/heartbeat.rs +0 -0
- {arrowbricks-3.0.3 → arrowbricks-3.0.4}/rust/arrowbricks_core/src/json_convert.rs +0 -0
- {arrowbricks-3.0.3 → arrowbricks-3.0.4}/rust/arrowbricks_core/src/lib.rs +0 -0
- {arrowbricks-3.0.3 → arrowbricks-3.0.4}/rust/arrowbricks_core/src/thrift.rs +0 -0
- {arrowbricks-3.0.3 → arrowbricks-3.0.4}/rust/arrowbricks_core/tests/wiremock_pipeline.rs +0 -0
- {arrowbricks-3.0.3 → arrowbricks-3.0.4}/rust/arrowbricks_core/tests/wiremock_thrift.rs +0 -0
- {arrowbricks-3.0.3 → arrowbricks-3.0.4}/rust/arrowbricks_core/tests/wiremock_volume_files.rs +0 -0
- {arrowbricks-3.0.3 → arrowbricks-3.0.4}/rust/arrowbricks_core/tests_py/conftest.py +0 -0
- {arrowbricks-3.0.3 → arrowbricks-3.0.4}/rust/arrowbricks_core/tests_py/test_ipc_stream.py +0 -0
- {arrowbricks-3.0.3 → arrowbricks-3.0.4}/rust/arrowbricks_core/tests_py/test_parameters.py +0 -0
- {arrowbricks-3.0.3 → arrowbricks-3.0.4}/rust/arrowbricks_core/tests_py/test_stream_ndjson_lines.py +0 -0
- {arrowbricks-3.0.3 → arrowbricks-3.0.4}/rust/arrowbricks_core/tests_py/test_streaming.py +0 -0
- {arrowbricks-3.0.3 → arrowbricks-3.0.4}/rust/arrowbricks_core/tests_py/test_thrift.py +0 -0
- {arrowbricks-3.0.3 → arrowbricks-3.0.4}/rust/arrowbricks_core/tests_py/test_token_provider.py +0 -0
- {arrowbricks-3.0.3 → arrowbricks-3.0.4}/rust/arrowbricks_core/tests_py/test_volume_files.py +0 -0
- {arrowbricks-3.0.3 → arrowbricks-3.0.4}/rust/arrowbricks_core/tests_py/thrift_mock.py +0 -0
- {arrowbricks-3.0.3 → arrowbricks-3.0.4}/src/arrowbricks/__init__.py +0 -0
- {arrowbricks-3.0.3 → arrowbricks-3.0.4}/src/arrowbricks/_core.pyi +0 -0
- {arrowbricks-3.0.3 → arrowbricks-3.0.4}/src/arrowbricks/_streaming.py +0 -0
- {arrowbricks-3.0.3 → arrowbricks-3.0.4}/src/arrowbricks/cursor.py +0 -0
- {arrowbricks-3.0.3 → arrowbricks-3.0.4}/src/arrowbricks/py.typed +0 -0
|
@@ -1,6 +1,6 @@
|
|
|
1
1
|
[project]
|
|
2
2
|
name = "arrowbricks"
|
|
3
|
-
version = "3.0.
|
|
3
|
+
version = "3.0.4"
|
|
4
4
|
description = "Runs SQL against a Databricks SQL warehouse via the Statement Execution API and hands you the result as Arrow -- a DB-API-ish Cursor (fetchone/fetchmany/fetchall/fetchall_arrow) or NDJSON streaming. Rust/PyO3 core throughout -- zero required runtime dependencies."
|
|
5
5
|
readme = "README.md"
|
|
6
6
|
license = "MIT"
|
|
@@ -738,20 +738,41 @@ impl DbClient {
|
|
|
738
738
|
// Python's DatabricksClient defaults to 6, tuned for asyncio+GIL
|
|
739
739
|
// where higher concurrency stops paying off past single digits
|
|
740
740
|
// (see its own comment). This Rust core's real OS-thread
|
|
741
|
-
// parallelism keeps paying off well past that -- measured
|
|
742
|
-
// a
|
|
743
|
-
//
|
|
744
|
-
//
|
|
745
|
-
//
|
|
746
|
-
//
|
|
747
|
-
//
|
|
748
|
-
//
|
|
749
|
-
//
|
|
750
|
-
//
|
|
751
|
-
//
|
|
752
|
-
//
|
|
753
|
-
//
|
|
754
|
-
//
|
|
741
|
+
// parallelism keeps paying off well past that -- measured
|
|
742
|
+
// against a 400-chunk/5.6M-row/120-column table (16=140s,
|
|
743
|
+
// 32=~113s avg of 3, 64=114s, 96=~102s avg of 2, 128=122s). 64
|
|
744
|
+
// was picked as a safe middle-ground bump with no observed
|
|
745
|
+
// downside on real data, not a claim that it's the true optimum.
|
|
746
|
+
//
|
|
747
|
+
// Re-attempted 2026-08-09 after switching away from http2/
|
|
748
|
+
// aws-lc-rs to ring, on the theory that either transport change
|
|
749
|
+
// could have moved the optimum: a first pass reported 96 as
|
|
750
|
+
// ~13-16% faster than 64 on both this table and a second,
|
|
751
|
+
// larger one (fact_sales_order_invoiced, 20M rows/295 cols) --
|
|
752
|
+
// **found on review to be a false positive**. The benchmark
|
|
753
|
+
// script called `connect()`/`arrowbricks.connect()` without
|
|
754
|
+
// ever passing `chunk_fetch_concurrency=` explicitly, so every
|
|
755
|
+
// "level" it claimed to test actually ran at whatever
|
|
756
|
+
// `PyDbClient::new`'s own `#[pyo3(signature = ...)]` default
|
|
757
|
+
// was (unchanged at 64 throughout, since only this Rust-side
|
|
758
|
+
// constant had been edited, and it's overridden unconditionally
|
|
759
|
+
// by `PyDbClient::new`'s explicit `.with_concurrency(...)` call
|
|
760
|
+
// for every real Python caller regardless of this constant's
|
|
761
|
+
// value) -- i.e. every "64 vs 96 vs 128" comparison was actually
|
|
762
|
+
// 64 vs 64 vs 64, and the reported gap was warehouse/network
|
|
763
|
+
// run-to-run noise, not a code effect. Caught by re-running a
|
|
764
|
+
// controlled, interleaved A/B (`chunk_fetch_concurrency=`
|
|
765
|
+
// passed explicitly each time, no rebuild needed) on
|
|
766
|
+
// `dim_article`: 64/96/64/96 measured 125.60s/129.92s/137.50s/
|
|
767
|
+
// 134.30s -- no consistent winner, well within run-to-run
|
|
768
|
+
// noise. Reverted to 64. If re-attempting this again, always
|
|
769
|
+
// pass `chunk_fetch_concurrency=` explicitly in the benchmark
|
|
770
|
+
// script itself rather than relying on rebuilding with a
|
|
771
|
+
// different default -- this exact mistake is easy to repeat
|
|
772
|
+
// otherwise, since the four independent places this default is
|
|
773
|
+
// hardcoded (see AGENTS.md) make "I changed the constant" and "a
|
|
774
|
+
// real Python caller now uses that constant" two different,
|
|
775
|
+
// easily-conflated claims.
|
|
755
776
|
chunk_fetch_concurrency: DEFAULT_CHUNK_FETCH_CONCURRENCY,
|
|
756
777
|
warehouse_start_timeout: Duration::from_secs(300),
|
|
757
778
|
warehouse_confirmed_running_ttl: Duration::from_secs(30),
|
|
@@ -1008,7 +1008,24 @@ fn build_inline_blob(
|
|
|
1008
1008
|
batches: Vec<thrift::ArrowBatch>,
|
|
1009
1009
|
lz4_compressed: bool,
|
|
1010
1010
|
) -> Result<(bytes::Bytes, i64), ApiError> {
|
|
1011
|
-
|
|
1011
|
+
// Pre-size the output buffer to avoid repeated allocations. Schema is
|
|
1012
|
+
// typically a few KB; batches are typically 10s-100s of KB uncompressed.
|
|
1013
|
+
// For compressed data, use the same * 4 heuristic as decompress_lz4_frame
|
|
1014
|
+
// (Arrow IPC data typically compresses several-fold); for uncompressed,
|
|
1015
|
+
// the exact sizes are known.
|
|
1016
|
+
let schema_size = schema_bytes.as_ref().map(|s| s.len()).unwrap_or(0);
|
|
1017
|
+
let batches_size: usize = batches
|
|
1018
|
+
.iter()
|
|
1019
|
+
.map(|b| {
|
|
1020
|
+
if lz4_compressed {
|
|
1021
|
+
b.batch.len() * 4
|
|
1022
|
+
} else {
|
|
1023
|
+
b.batch.len()
|
|
1024
|
+
}
|
|
1025
|
+
})
|
|
1026
|
+
.sum();
|
|
1027
|
+
let mut out = Vec::with_capacity(schema_size + batches_size);
|
|
1028
|
+
|
|
1012
1029
|
if let Some(s) = &schema_bytes {
|
|
1013
1030
|
out.extend_from_slice(s);
|
|
1014
1031
|
}
|
|
@@ -1016,8 +1033,16 @@ fn build_inline_blob(
|
|
|
1016
1033
|
for b in batches {
|
|
1017
1034
|
row_count += b.row_count;
|
|
1018
1035
|
if lz4_compressed {
|
|
1019
|
-
|
|
1020
|
-
|
|
1036
|
+
// Decompress directly into the output buffer using FrameDecoder,
|
|
1037
|
+
// avoiding a separate intermediate allocation and copy. The
|
|
1038
|
+
// FrameDecoder's read_to_end appends to the existing buffer.
|
|
1039
|
+
use std::io::Read;
|
|
1040
|
+
let mut decoder = lz4_flex::frame::FrameDecoder::new(&b.batch[..]);
|
|
1041
|
+
while !decoder.get_ref().is_empty() {
|
|
1042
|
+
decoder
|
|
1043
|
+
.read_to_end(&mut out)
|
|
1044
|
+
.map_err(|e| ApiError::permanent(format!("LZ4 frame decompress failed: {e}")))?;
|
|
1045
|
+
}
|
|
1021
1046
|
} else {
|
|
1022
1047
|
out.extend_from_slice(&b.batch);
|
|
1023
1048
|
}
|
|
@@ -50,7 +50,9 @@ class DatabricksClient:
|
|
|
50
50
|
# chunk's raw bytes in memory. 64 -- tuned for the Rust core's real
|
|
51
51
|
# OS-thread parallelism, which scales well past what asyncio+GIL
|
|
52
52
|
# concurrency used to buy; measured against a real 400-chunk/
|
|
53
|
-
# 5.6M-row table (see client.rs's own comment for the numbers
|
|
53
|
+
# 5.6M-row table (see client.rs's own comment for the numbers, and
|
|
54
|
+
# for a since-reverted attempt to bump this further that turned out
|
|
55
|
+
# to be a benchmarking false positive).
|
|
54
56
|
# compress_results: requests LZ4-compressed cloud-fetch chunks
|
|
55
57
|
# (matches databricks-sql-connector's own
|
|
56
58
|
# enable_query_result_lz4_compression default) -- measured ~2x
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
{arrowbricks-3.0.3 → arrowbricks-3.0.4}/rust/arrowbricks_core/tests/wiremock_volume_files.rs
RENAMED
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
{arrowbricks-3.0.3 → arrowbricks-3.0.4}/rust/arrowbricks_core/tests_py/test_stream_ndjson_lines.py
RENAMED
|
File without changes
|
|
File without changes
|
|
File without changes
|
{arrowbricks-3.0.3 → arrowbricks-3.0.4}/rust/arrowbricks_core/tests_py/test_token_provider.py
RENAMED
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|