arrowbricks 1.3.0__tar.gz → 1.3.2__tar.gz
This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
- {arrowbricks-1.3.0 → arrowbricks-1.3.2}/PKG-INFO +1 -1
- {arrowbricks-1.3.0 → arrowbricks-1.3.2}/pyproject.toml +1 -1
- {arrowbricks-1.3.0 → arrowbricks-1.3.2}/rust/arrowbricks_core/Cargo.lock +1 -1
- {arrowbricks-1.3.0 → arrowbricks-1.3.2}/rust/arrowbricks_core/Cargo.toml +1 -1
- {arrowbricks-1.3.0 → arrowbricks-1.3.2}/rust/arrowbricks_core/src/pipeline.rs +241 -7
- {arrowbricks-1.3.0 → arrowbricks-1.3.2}/LICENSE +0 -0
- {arrowbricks-1.3.0 → arrowbricks-1.3.2}/README.md +0 -0
- {arrowbricks-1.3.0 → arrowbricks-1.3.2}/rust/arrowbricks_core/.gitignore +0 -0
- {arrowbricks-1.3.0 → arrowbricks-1.3.2}/rust/arrowbricks_core/README.md +0 -0
- {arrowbricks-1.3.0 → arrowbricks-1.3.2}/rust/arrowbricks_core/examples/duckdb_query.py +0 -0
- {arrowbricks-1.3.0 → arrowbricks-1.3.2}/rust/arrowbricks_core/examples/fastapi_sse.py +0 -0
- {arrowbricks-1.3.0 → arrowbricks-1.3.2}/rust/arrowbricks_core/rustfmt.toml +0 -0
- {arrowbricks-1.3.0 → arrowbricks-1.3.2}/rust/arrowbricks_core/src/client.rs +0 -0
- {arrowbricks-1.3.0 → arrowbricks-1.3.2}/rust/arrowbricks_core/src/heartbeat.rs +0 -0
- {arrowbricks-1.3.0 → arrowbricks-1.3.2}/rust/arrowbricks_core/src/lib.rs +0 -0
- {arrowbricks-1.3.0 → arrowbricks-1.3.2}/rust/arrowbricks_core/tests/wiremock_pipeline.rs +0 -0
- {arrowbricks-1.3.0 → arrowbricks-1.3.2}/rust/arrowbricks_core/tests/wiremock_volume_files.rs +0 -0
- {arrowbricks-1.3.0 → arrowbricks-1.3.2}/rust/arrowbricks_core/tests_py/test_execute_json.py +0 -0
- {arrowbricks-1.3.0 → arrowbricks-1.3.2}/rust/arrowbricks_core/tests_py/test_ipc_stream.py +0 -0
- {arrowbricks-1.3.0 → arrowbricks-1.3.2}/rust/arrowbricks_core/tests_py/test_parameters.py +0 -0
- {arrowbricks-1.3.0 → arrowbricks-1.3.2}/rust/arrowbricks_core/tests_py/test_stream_ndjson_lines.py +0 -0
- {arrowbricks-1.3.0 → arrowbricks-1.3.2}/rust/arrowbricks_core/tests_py/test_streaming.py +0 -0
- {arrowbricks-1.3.0 → arrowbricks-1.3.2}/rust/arrowbricks_core/tests_py/test_token_provider.py +0 -0
- {arrowbricks-1.3.0 → arrowbricks-1.3.2}/rust/arrowbricks_core/tests_py/test_volume_files.py +0 -0
- {arrowbricks-1.3.0 → arrowbricks-1.3.2}/src/arrowbricks/__init__.py +0 -0
- {arrowbricks-1.3.0 → arrowbricks-1.3.2}/src/arrowbricks/_core.pyi +0 -0
- {arrowbricks-1.3.0 → arrowbricks-1.3.2}/src/arrowbricks/_streaming.py +0 -0
- {arrowbricks-1.3.0 → arrowbricks-1.3.2}/src/arrowbricks/client.py +0 -0
- {arrowbricks-1.3.0 → arrowbricks-1.3.2}/src/arrowbricks/cursor.py +0 -0
- {arrowbricks-1.3.0 → arrowbricks-1.3.2}/src/arrowbricks/py.typed +0 -0
|
@@ -1,6 +1,6 @@
|
|
|
1
1
|
[project]
|
|
2
2
|
name = "arrowbricks"
|
|
3
|
-
version = "1.3.
|
|
3
|
+
version = "1.3.2"
|
|
4
4
|
description = "Runs SQL against a Databricks SQL warehouse via the Statement Execution API and hands you the result as Arrow -- a DB-API-ish Cursor (fetchone/fetchmany/fetchall/fetchall_arrow) or NDJSON streaming. Rust/PyO3 core throughout -- zero required runtime dependencies."
|
|
5
5
|
readme = "README.md"
|
|
6
6
|
license = "MIT"
|
|
@@ -2,12 +2,12 @@
|
|
|
2
2
|
//! decode of the reordered chunk stream via arrow-rs.
|
|
3
3
|
|
|
4
4
|
use std::collections::{HashMap, VecDeque};
|
|
5
|
-
use std::io::Cursor as IoCursor;
|
|
6
5
|
use std::sync::Arc;
|
|
7
6
|
|
|
8
7
|
use arrow::array::{Array, AsArray};
|
|
8
|
+
use arrow::buffer::Buffer as ArrowBuffer;
|
|
9
9
|
use arrow::datatypes::{DataType, Float32Type, Float64Type, SchemaRef};
|
|
10
|
-
use arrow::ipc::reader::
|
|
10
|
+
use arrow::ipc::reader::StreamDecoder;
|
|
11
11
|
use arrow::record_batch::RecordBatch;
|
|
12
12
|
use bytes::Bytes;
|
|
13
13
|
use serde_json::Value;
|
|
@@ -97,15 +97,66 @@ impl ExecuteResult {
|
|
|
97
97
|
}
|
|
98
98
|
}
|
|
99
99
|
|
|
100
|
+
/// Decodes a chunk's raw Arrow-IPC stream bytes into batches. Uses
|
|
101
|
+
/// `StreamDecoder`'s push-based interface fed by an `arrow::buffer::Buffer`
|
|
102
|
+
/// built directly from `blob` (`Buffer::from(bytes::Bytes)`, confirmed
|
|
103
|
+
/// zero-copy in arrow-buffer's own source -- `bytes.rs`'s
|
|
104
|
+
/// `impl From<bytes::Bytes> for Bytes` stores the original `bytes::Bytes` via
|
|
105
|
+
/// `Deallocation::Custom`, no memcpy) instead of the higher-level
|
|
106
|
+
/// `StreamReader` (reads via `std::io::Read` into freshly allocated buffers,
|
|
107
|
+
/// copying every column's data out of `blob` on every decode -- what this
|
|
108
|
+
/// used before). For properly aligned IPC data (the normal case -- Databricks
|
|
109
|
+
/// writes it, not this crate), decoded batches now slice directly into the
|
|
110
|
+
/// same allocation `blob` already held since the network fetch, cutting out
|
|
111
|
+
/// a second full copy of every chunk's bytes; `require_alignment` stays at
|
|
112
|
+
/// its default `false`, so a misaligned *fixed-width* buffer still falls back
|
|
113
|
+
/// to a copy automatically rather than erroring (arrow-ipc's own documented
|
|
114
|
+
/// behavior) -- variable-width values, null bitmaps, and nested/dictionary
|
|
115
|
+
/// children all stay zero-copy regardless. One exception this doesn't cover:
|
|
116
|
+
/// if a `RecordBatch` message declared IPC *buffer*-level compression (a
|
|
117
|
+
/// different, unrelated feature from this crate's own cloud-fetch
|
|
118
|
+
/// `result_compression` unwrap in `client.rs`, which already ran before this
|
|
119
|
+
/// function ever sees the bytes), `arrow-ipc`'s own reader always
|
|
120
|
+
/// decompresses into fresh buffers there -- not something Databricks has
|
|
121
|
+
/// been observed to use in this format, but not something this crate
|
|
122
|
+
/// controls either.
|
|
123
|
+
///
|
|
124
|
+
/// A zero-length `blob` is rejected explicitly rather than handed to
|
|
125
|
+
/// `StreamDecoder`: found in code review that an empty buffer makes the
|
|
126
|
+
/// `while` loop below a no-op and `decoder.finish()` sees a still-pristine
|
|
127
|
+
/// decoder state, which its own `Ok(())` arm treats as a *clean, empty*
|
|
128
|
+
/// stream -- silently returning zero batches with no error at all, the same
|
|
129
|
+
/// silent-truncation failure mode as the real multi-frame LZ4 bug this crate
|
|
130
|
+
/// already shipped once (see the `result_compression` invariant above). The
|
|
131
|
+
/// old `StreamReader`-based version failed loudly on this input instead
|
|
132
|
+
/// ("Expected schema message, found empty stream"); this restores that.
|
|
100
133
|
fn decode_chunk(blob: &Bytes) -> Result<Vec<RecordBatch>, ApiError> {
|
|
101
|
-
|
|
134
|
+
if blob.is_empty() {
|
|
135
|
+
return Err(ApiError {
|
|
136
|
+
message: "empty Arrow IPC chunk: expected at least a schema message".to_string(),
|
|
137
|
+
transient: false,
|
|
138
|
+
});
|
|
139
|
+
}
|
|
140
|
+
let mut buffer = ArrowBuffer::from(blob.clone());
|
|
141
|
+
let mut decoder = StreamDecoder::new();
|
|
142
|
+
let mut batches = Vec::new();
|
|
143
|
+
while !buffer.is_empty() {
|
|
144
|
+
match decoder.decode(&mut buffer) {
|
|
145
|
+
Ok(Some(batch)) => batches.push(batch),
|
|
146
|
+
Ok(None) => {}
|
|
147
|
+
Err(e) => {
|
|
148
|
+
return Err(ApiError {
|
|
149
|
+
message: format!("Arrow IPC decode error: {e}"),
|
|
150
|
+
transient: false,
|
|
151
|
+
});
|
|
152
|
+
}
|
|
153
|
+
}
|
|
154
|
+
}
|
|
155
|
+
decoder.finish().map_err(|e| ApiError {
|
|
102
156
|
message: format!("bad Arrow IPC stream: {e}"),
|
|
103
157
|
transient: false,
|
|
104
158
|
})?;
|
|
105
|
-
|
|
106
|
-
message: format!("Arrow IPC decode error: {e}"),
|
|
107
|
-
transient: false,
|
|
108
|
-
})
|
|
159
|
+
Ok(batches)
|
|
109
160
|
}
|
|
110
161
|
|
|
111
162
|
/// Caps how many chunks `ResultStream::fetch_at_least` will pull/decode
|
|
@@ -797,4 +848,187 @@ mod tests {
|
|
|
797
848
|
"a real NULL must stay null, never become a string"
|
|
798
849
|
);
|
|
799
850
|
}
|
|
851
|
+
|
|
852
|
+
/// Regression test for the switch from `StreamReader` to the push-based
|
|
853
|
+
/// `StreamDecoder` (see `decode_chunk`'s doc comment): a single chunk can
|
|
854
|
+
/// contain more than one Arrow-IPC `RecordBatch` message back to back in
|
|
855
|
+
/// the same stream, and `StreamDecoder::decode` only ever returns one
|
|
856
|
+
/// batch per call -- `decode_chunk` must keep calling it until the whole
|
|
857
|
+
/// buffer is drained, not stop after the first. `StreamReader`'s own
|
|
858
|
+
/// `Iterator` impl made this automatic; the lower-level API doesn't, so
|
|
859
|
+
/// this is exactly the kind of thing that regresses silently (a single-
|
|
860
|
+
/// batch-per-chunk bug would still pass every other test in this suite,
|
|
861
|
+
/// since none of them writes more than one batch per chunk).
|
|
862
|
+
#[test]
|
|
863
|
+
fn decode_chunk_reads_every_record_batch_in_a_multi_batch_stream() {
|
|
864
|
+
use arrow::ipc::writer::StreamWriter;
|
|
865
|
+
|
|
866
|
+
let batch_a = make_batch(vec![1, 2], vec![1.0, 2.0]);
|
|
867
|
+
let batch_b = make_batch(vec![3, 4, 5], vec![3.0, 4.0, 5.0]);
|
|
868
|
+
|
|
869
|
+
let mut buf = Vec::new();
|
|
870
|
+
{
|
|
871
|
+
let mut writer = StreamWriter::try_new(&mut buf, &batch_a.schema()).unwrap();
|
|
872
|
+
writer.write(&batch_a).unwrap();
|
|
873
|
+
writer.write(&batch_b).unwrap();
|
|
874
|
+
writer.finish().unwrap();
|
|
875
|
+
}
|
|
876
|
+
|
|
877
|
+
let batches = decode_chunk(&Bytes::from(buf)).unwrap();
|
|
878
|
+
assert_eq!(
|
|
879
|
+
batches.len(),
|
|
880
|
+
2,
|
|
881
|
+
"both record batches in the stream must be decoded, not just the first"
|
|
882
|
+
);
|
|
883
|
+
let total_rows: usize = batches.iter().map(|b| b.num_rows()).sum();
|
|
884
|
+
assert_eq!(total_rows, 5);
|
|
885
|
+
}
|
|
886
|
+
|
|
887
|
+
/// Regression test for a bug caught in code review before it shipped: an
|
|
888
|
+
/// empty (zero-byte) blob made `StreamDecoder`'s decode loop a no-op and
|
|
889
|
+
/// `decoder.finish()` saw a still-pristine state, which it treats as a
|
|
890
|
+
/// clean empty stream -- silently returning zero batches with no error at
|
|
891
|
+
/// all, instead of the loud failure the old `StreamReader`-based version
|
|
892
|
+
/// gave on the same input ("Expected schema message, found empty
|
|
893
|
+
/// stream"). Same silent-truncation shape as the real multi-frame LZ4 bug
|
|
894
|
+
/// this crate already shipped once (see `client.rs`'s
|
|
895
|
+
/// `decompress_lz4_frame` doc comment) -- a genuinely empty chunk blob
|
|
896
|
+
/// must never be mistaken for "legitimately zero rows".
|
|
897
|
+
#[test]
|
|
898
|
+
fn decode_chunk_rejects_an_empty_blob() {
|
|
899
|
+
let err = decode_chunk(&Bytes::new()).expect_err("an empty blob must error, not silently decode to zero rows");
|
|
900
|
+
assert!(
|
|
901
|
+
err.message.contains("empty"),
|
|
902
|
+
"error should mention the blob was empty: {}",
|
|
903
|
+
err.message
|
|
904
|
+
);
|
|
905
|
+
}
|
|
906
|
+
|
|
907
|
+
/// Companion to `decode_chunk_rejects_an_empty_blob` -- proves the empty-
|
|
908
|
+
/// blob check doesn't overcorrect: a *non-empty* stream containing only a
|
|
909
|
+
/// schema message and no `RecordBatch` at all (a legitimate shape for a
|
|
910
|
+
/// genuinely empty query result) must still decode successfully to zero
|
|
911
|
+
/// batches, not error.
|
|
912
|
+
#[test]
|
|
913
|
+
fn decode_chunk_accepts_a_schema_only_stream_with_zero_batches() {
|
|
914
|
+
use arrow::datatypes::{Field, Schema};
|
|
915
|
+
use arrow::ipc::writer::StreamWriter;
|
|
916
|
+
|
|
917
|
+
let schema = Arc::new(Schema::new(vec![Field::new("id", DataType::Int64, false)]));
|
|
918
|
+
let mut buf = Vec::new();
|
|
919
|
+
{
|
|
920
|
+
let mut writer = StreamWriter::try_new(&mut buf, &schema).unwrap();
|
|
921
|
+
writer.finish().unwrap();
|
|
922
|
+
}
|
|
923
|
+
|
|
924
|
+
let batches = decode_chunk(&Bytes::from(buf)).unwrap();
|
|
925
|
+
assert_eq!(
|
|
926
|
+
batches.len(),
|
|
927
|
+
0,
|
|
928
|
+
"a schema-only stream with no batches is valid, not an error"
|
|
929
|
+
);
|
|
930
|
+
}
|
|
931
|
+
|
|
932
|
+
/// Regression/documentation test for a real behavior change found in code
|
|
933
|
+
/// review: `StreamDecoder` (unlike the old `StreamReader`) hard-errors on
|
|
934
|
+
/// any bytes left over after a stream's own EOS marker, instead of
|
|
935
|
+
/// silently ignoring them. Locking this in deliberately -- erroring beats
|
|
936
|
+
/// silently dropping whatever came after the truncation point, same
|
|
937
|
+
/// reasoning as the empty-blob check above -- even though real Databricks
|
|
938
|
+
/// chunks have not been observed to have trailing bytes.
|
|
939
|
+
#[test]
|
|
940
|
+
fn decode_chunk_errors_on_trailing_bytes_after_a_complete_stream() {
|
|
941
|
+
use arrow::ipc::writer::StreamWriter;
|
|
942
|
+
|
|
943
|
+
let batch = make_batch(vec![1, 2], vec![1.0, 2.0]);
|
|
944
|
+
let mut buf = Vec::new();
|
|
945
|
+
{
|
|
946
|
+
let mut writer = StreamWriter::try_new(&mut buf, &batch.schema()).unwrap();
|
|
947
|
+
writer.write(&batch).unwrap();
|
|
948
|
+
writer.finish().unwrap();
|
|
949
|
+
}
|
|
950
|
+
buf.extend_from_slice(&[0xAA; 8]);
|
|
951
|
+
|
|
952
|
+
decode_chunk(&Bytes::from(buf)).expect_err("trailing bytes after a complete stream's EOS marker must error");
|
|
953
|
+
}
|
|
954
|
+
|
|
955
|
+
/// Diagnostic only, not a correctness check (relative timing is too
|
|
956
|
+
/// flaky for CI) -- `cargo test --release -- --ignored --nocapture
|
|
957
|
+
/// decode_chunk_speed` to compare the current `StreamDecoder`-based
|
|
958
|
+
/// `decode_chunk` against the old `StreamReader`-based approach it
|
|
959
|
+
/// replaced, on a batch shaped like a real chunk (120 columns, mixed
|
|
960
|
+
/// types, 50k rows -- this session's own real-table benchmark).
|
|
961
|
+
#[test]
|
|
962
|
+
#[ignore]
|
|
963
|
+
fn decode_chunk_speed_vs_stream_reader() {
|
|
964
|
+
use arrow::array::{Float64Array, Int64Array, StringArray};
|
|
965
|
+
use arrow::datatypes::{Field, Schema};
|
|
966
|
+
use arrow::ipc::writer::StreamWriter;
|
|
967
|
+
use std::io::Cursor as IoCursor;
|
|
968
|
+
use std::time::Instant;
|
|
969
|
+
|
|
970
|
+
const ROWS: usize = 50_000;
|
|
971
|
+
const COLS: usize = 120;
|
|
972
|
+
|
|
973
|
+
let mut fields = Vec::with_capacity(COLS);
|
|
974
|
+
let mut columns: Vec<Arc<dyn Array>> = Vec::with_capacity(COLS);
|
|
975
|
+
for i in 0..COLS {
|
|
976
|
+
match i % 3 {
|
|
977
|
+
0 => {
|
|
978
|
+
fields.push(Field::new(format!("c{i}"), DataType::Int64, false));
|
|
979
|
+
columns.push(Arc::new(Int64Array::from((0..ROWS as i64).collect::<Vec<_>>())));
|
|
980
|
+
}
|
|
981
|
+
1 => {
|
|
982
|
+
fields.push(Field::new(format!("c{i}"), DataType::Float64, false));
|
|
983
|
+
columns.push(Arc::new(Float64Array::from(
|
|
984
|
+
(0..ROWS).map(|r| r as f64 * 1.5).collect::<Vec<_>>(),
|
|
985
|
+
)));
|
|
986
|
+
}
|
|
987
|
+
_ => {
|
|
988
|
+
fields.push(Field::new(format!("c{i}"), DataType::Utf8, false));
|
|
989
|
+
columns.push(Arc::new(StringArray::from(
|
|
990
|
+
(0..ROWS).map(|r| format!("row-{r}")).collect::<Vec<_>>(),
|
|
991
|
+
)));
|
|
992
|
+
}
|
|
993
|
+
}
|
|
994
|
+
}
|
|
995
|
+
let schema = Arc::new(Schema::new(fields));
|
|
996
|
+
let batch = RecordBatch::try_new(schema.clone(), columns).unwrap();
|
|
997
|
+
|
|
998
|
+
let mut buf = Vec::new();
|
|
999
|
+
{
|
|
1000
|
+
let mut writer = StreamWriter::try_new(&mut buf, &schema).unwrap();
|
|
1001
|
+
writer.write(&batch).unwrap();
|
|
1002
|
+
writer.finish().unwrap();
|
|
1003
|
+
}
|
|
1004
|
+
let blob = Bytes::from(buf);
|
|
1005
|
+
|
|
1006
|
+
const ITERS: u32 = 30;
|
|
1007
|
+
|
|
1008
|
+
// Old approach: StreamReader over an IoCursor -- copies every
|
|
1009
|
+
// column's data into freshly allocated buffers on every decode.
|
|
1010
|
+
let old_start = Instant::now();
|
|
1011
|
+
for _ in 0..ITERS {
|
|
1012
|
+
let reader = arrow::ipc::reader::StreamReader::try_new(IoCursor::new(&blob[..]), None).unwrap();
|
|
1013
|
+
let batches: Vec<RecordBatch> = reader.collect::<Result<Vec<_>, _>>().unwrap();
|
|
1014
|
+
assert_eq!(batches[0].num_rows(), ROWS);
|
|
1015
|
+
}
|
|
1016
|
+
let old_elapsed = old_start.elapsed();
|
|
1017
|
+
|
|
1018
|
+
// New approach: this file's actual decode_chunk.
|
|
1019
|
+
let new_start = Instant::now();
|
|
1020
|
+
for _ in 0..ITERS {
|
|
1021
|
+
let batches = decode_chunk(&blob).unwrap();
|
|
1022
|
+
assert_eq!(batches[0].num_rows(), ROWS);
|
|
1023
|
+
}
|
|
1024
|
+
let new_elapsed = new_start.elapsed();
|
|
1025
|
+
|
|
1026
|
+
println!(
|
|
1027
|
+
"decode_chunk speed ({COLS} cols x {ROWS} rows, {ITERS} iters): \
|
|
1028
|
+
StreamReader (old) = {old_elapsed:?} ({:?}/iter), \
|
|
1029
|
+
StreamDecoder (new) = {new_elapsed:?} ({:?}/iter)",
|
|
1030
|
+
old_elapsed / ITERS,
|
|
1031
|
+
new_elapsed / ITERS,
|
|
1032
|
+
);
|
|
1033
|
+
}
|
|
800
1034
|
}
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
{arrowbricks-1.3.0 → arrowbricks-1.3.2}/rust/arrowbricks_core/tests/wiremock_volume_files.rs
RENAMED
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
{arrowbricks-1.3.0 → arrowbricks-1.3.2}/rust/arrowbricks_core/tests_py/test_stream_ndjson_lines.py
RENAMED
|
File without changes
|
|
File without changes
|
{arrowbricks-1.3.0 → arrowbricks-1.3.2}/rust/arrowbricks_core/tests_py/test_token_provider.py
RENAMED
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|