arrowbricks 1.3.0__tar.gz → 1.3.2__tar.gz

This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
Files changed (30) hide show
  1. {arrowbricks-1.3.0 → arrowbricks-1.3.2}/PKG-INFO +1 -1
  2. {arrowbricks-1.3.0 → arrowbricks-1.3.2}/pyproject.toml +1 -1
  3. {arrowbricks-1.3.0 → arrowbricks-1.3.2}/rust/arrowbricks_core/Cargo.lock +1 -1
  4. {arrowbricks-1.3.0 → arrowbricks-1.3.2}/rust/arrowbricks_core/Cargo.toml +1 -1
  5. {arrowbricks-1.3.0 → arrowbricks-1.3.2}/rust/arrowbricks_core/src/pipeline.rs +241 -7
  6. {arrowbricks-1.3.0 → arrowbricks-1.3.2}/LICENSE +0 -0
  7. {arrowbricks-1.3.0 → arrowbricks-1.3.2}/README.md +0 -0
  8. {arrowbricks-1.3.0 → arrowbricks-1.3.2}/rust/arrowbricks_core/.gitignore +0 -0
  9. {arrowbricks-1.3.0 → arrowbricks-1.3.2}/rust/arrowbricks_core/README.md +0 -0
  10. {arrowbricks-1.3.0 → arrowbricks-1.3.2}/rust/arrowbricks_core/examples/duckdb_query.py +0 -0
  11. {arrowbricks-1.3.0 → arrowbricks-1.3.2}/rust/arrowbricks_core/examples/fastapi_sse.py +0 -0
  12. {arrowbricks-1.3.0 → arrowbricks-1.3.2}/rust/arrowbricks_core/rustfmt.toml +0 -0
  13. {arrowbricks-1.3.0 → arrowbricks-1.3.2}/rust/arrowbricks_core/src/client.rs +0 -0
  14. {arrowbricks-1.3.0 → arrowbricks-1.3.2}/rust/arrowbricks_core/src/heartbeat.rs +0 -0
  15. {arrowbricks-1.3.0 → arrowbricks-1.3.2}/rust/arrowbricks_core/src/lib.rs +0 -0
  16. {arrowbricks-1.3.0 → arrowbricks-1.3.2}/rust/arrowbricks_core/tests/wiremock_pipeline.rs +0 -0
  17. {arrowbricks-1.3.0 → arrowbricks-1.3.2}/rust/arrowbricks_core/tests/wiremock_volume_files.rs +0 -0
  18. {arrowbricks-1.3.0 → arrowbricks-1.3.2}/rust/arrowbricks_core/tests_py/test_execute_json.py +0 -0
  19. {arrowbricks-1.3.0 → arrowbricks-1.3.2}/rust/arrowbricks_core/tests_py/test_ipc_stream.py +0 -0
  20. {arrowbricks-1.3.0 → arrowbricks-1.3.2}/rust/arrowbricks_core/tests_py/test_parameters.py +0 -0
  21. {arrowbricks-1.3.0 → arrowbricks-1.3.2}/rust/arrowbricks_core/tests_py/test_stream_ndjson_lines.py +0 -0
  22. {arrowbricks-1.3.0 → arrowbricks-1.3.2}/rust/arrowbricks_core/tests_py/test_streaming.py +0 -0
  23. {arrowbricks-1.3.0 → arrowbricks-1.3.2}/rust/arrowbricks_core/tests_py/test_token_provider.py +0 -0
  24. {arrowbricks-1.3.0 → arrowbricks-1.3.2}/rust/arrowbricks_core/tests_py/test_volume_files.py +0 -0
  25. {arrowbricks-1.3.0 → arrowbricks-1.3.2}/src/arrowbricks/__init__.py +0 -0
  26. {arrowbricks-1.3.0 → arrowbricks-1.3.2}/src/arrowbricks/_core.pyi +0 -0
  27. {arrowbricks-1.3.0 → arrowbricks-1.3.2}/src/arrowbricks/_streaming.py +0 -0
  28. {arrowbricks-1.3.0 → arrowbricks-1.3.2}/src/arrowbricks/client.py +0 -0
  29. {arrowbricks-1.3.0 → arrowbricks-1.3.2}/src/arrowbricks/cursor.py +0 -0
  30. {arrowbricks-1.3.0 → arrowbricks-1.3.2}/src/arrowbricks/py.typed +0 -0
@@ -1,6 +1,6 @@
1
1
  Metadata-Version: 2.4
2
2
  Name: arrowbricks
3
- Version: 1.3.0
3
+ Version: 1.3.2
4
4
  Requires-Dist: arro3-core>=0.8 ; extra == 'arro3'
5
5
  Provides-Extra: arro3
6
6
  License-File: LICENSE
@@ -1,6 +1,6 @@
1
1
  [project]
2
2
  name = "arrowbricks"
3
- version = "1.3.0"
3
+ version = "1.3.2"
4
4
  description = "Runs SQL against a Databricks SQL warehouse via the Statement Execution API and hands you the result as Arrow -- a DB-API-ish Cursor (fetchone/fetchmany/fetchall/fetchall_arrow) or NDJSON streaming. Rust/PyO3 core throughout -- zero required runtime dependencies."
5
5
  readme = "README.md"
6
6
  license = "MIT"
@@ -240,7 +240,7 @@ dependencies = [
240
240
 
241
241
  [[package]]
242
242
  name = "arrowbricks_core"
243
- version = "1.3.0"
243
+ version = "1.3.2"
244
244
  dependencies = [
245
245
  "arrow",
246
246
  "arrow-json",
@@ -1,6 +1,6 @@
1
1
  [package]
2
2
  name = "arrowbricks_core"
3
- version = "1.3.0"
3
+ version = "1.3.2"
4
4
  edition = "2024"
5
5
  readme = "README.md"
6
6
 
@@ -2,12 +2,12 @@
2
2
  //! decode of the reordered chunk stream via arrow-rs.
3
3
 
4
4
  use std::collections::{HashMap, VecDeque};
5
- use std::io::Cursor as IoCursor;
6
5
  use std::sync::Arc;
7
6
 
8
7
  use arrow::array::{Array, AsArray};
8
+ use arrow::buffer::Buffer as ArrowBuffer;
9
9
  use arrow::datatypes::{DataType, Float32Type, Float64Type, SchemaRef};
10
- use arrow::ipc::reader::StreamReader;
10
+ use arrow::ipc::reader::StreamDecoder;
11
11
  use arrow::record_batch::RecordBatch;
12
12
  use bytes::Bytes;
13
13
  use serde_json::Value;
@@ -97,15 +97,66 @@ impl ExecuteResult {
97
97
  }
98
98
  }
99
99
 
100
+ /// Decodes a chunk's raw Arrow-IPC stream bytes into batches. Uses
101
+ /// `StreamDecoder`'s push-based interface fed by an `arrow::buffer::Buffer`
102
+ /// built directly from `blob` (`Buffer::from(bytes::Bytes)`, confirmed
103
+ /// zero-copy in arrow-buffer's own source -- `bytes.rs`'s
104
+ /// `impl From<bytes::Bytes> for Bytes` stores the original `bytes::Bytes` via
105
+ /// `Deallocation::Custom`, no memcpy) instead of the higher-level
106
+ /// `StreamReader` (reads via `std::io::Read` into freshly allocated buffers,
107
+ /// copying every column's data out of `blob` on every decode -- what this
108
+ /// used before). For properly aligned IPC data (the normal case -- Databricks
109
+ /// writes it, not this crate), decoded batches now slice directly into the
110
+ /// same allocation `blob` already held since the network fetch, cutting out
111
+ /// a second full copy of every chunk's bytes; `require_alignment` stays at
112
+ /// its default `false`, so a misaligned *fixed-width* buffer still falls back
113
+ /// to a copy automatically rather than erroring (arrow-ipc's own documented
114
+ /// behavior) -- variable-width values, null bitmaps, and nested/dictionary
115
+ /// children all stay zero-copy regardless. One exception this doesn't cover:
116
+ /// if a `RecordBatch` message declared IPC *buffer*-level compression (a
117
+ /// different, unrelated feature from this crate's own cloud-fetch
118
+ /// `result_compression` unwrap in `client.rs`, which already ran before this
119
+ /// function ever sees the bytes), `arrow-ipc`'s own reader always
120
+ /// decompresses into fresh buffers there -- not something Databricks has
121
+ /// been observed to use in this format, but not something this crate
122
+ /// controls either.
123
+ ///
124
+ /// A zero-length `blob` is rejected explicitly rather than handed to
125
+ /// `StreamDecoder`: found in code review that an empty buffer makes the
126
+ /// `while` loop below a no-op and `decoder.finish()` sees a still-pristine
127
+ /// decoder state, which its own `Ok(())` arm treats as a *clean, empty*
128
+ /// stream -- silently returning zero batches with no error at all, the same
129
+ /// silent-truncation failure mode as the real multi-frame LZ4 bug this crate
130
+ /// already shipped once (see the `result_compression` invariant above). The
131
+ /// old `StreamReader`-based version failed loudly on this input instead
132
+ /// ("Expected schema message, found empty stream"); this restores that.
100
133
  fn decode_chunk(blob: &Bytes) -> Result<Vec<RecordBatch>, ApiError> {
101
- let reader = StreamReader::try_new(IoCursor::new(&blob[..]), None).map_err(|e| ApiError {
134
+ if blob.is_empty() {
135
+ return Err(ApiError {
136
+ message: "empty Arrow IPC chunk: expected at least a schema message".to_string(),
137
+ transient: false,
138
+ });
139
+ }
140
+ let mut buffer = ArrowBuffer::from(blob.clone());
141
+ let mut decoder = StreamDecoder::new();
142
+ let mut batches = Vec::new();
143
+ while !buffer.is_empty() {
144
+ match decoder.decode(&mut buffer) {
145
+ Ok(Some(batch)) => batches.push(batch),
146
+ Ok(None) => {}
147
+ Err(e) => {
148
+ return Err(ApiError {
149
+ message: format!("Arrow IPC decode error: {e}"),
150
+ transient: false,
151
+ });
152
+ }
153
+ }
154
+ }
155
+ decoder.finish().map_err(|e| ApiError {
102
156
  message: format!("bad Arrow IPC stream: {e}"),
103
157
  transient: false,
104
158
  })?;
105
- reader.collect::<Result<Vec<_>, _>>().map_err(|e| ApiError {
106
- message: format!("Arrow IPC decode error: {e}"),
107
- transient: false,
108
- })
159
+ Ok(batches)
109
160
  }
110
161
 
111
162
  /// Caps how many chunks `ResultStream::fetch_at_least` will pull/decode
@@ -797,4 +848,187 @@ mod tests {
797
848
  "a real NULL must stay null, never become a string"
798
849
  );
799
850
  }
851
+
852
+ /// Regression test for the switch from `StreamReader` to the push-based
853
+ /// `StreamDecoder` (see `decode_chunk`'s doc comment): a single chunk can
854
+ /// contain more than one Arrow-IPC `RecordBatch` message back to back in
855
+ /// the same stream, and `StreamDecoder::decode` only ever returns one
856
+ /// batch per call -- `decode_chunk` must keep calling it until the whole
857
+ /// buffer is drained, not stop after the first. `StreamReader`'s own
858
+ /// `Iterator` impl made this automatic; the lower-level API doesn't, so
859
+ /// this is exactly the kind of thing that regresses silently (a single-
860
+ /// batch-per-chunk bug would still pass every other test in this suite,
861
+ /// since none of them writes more than one batch per chunk).
862
+ #[test]
863
+ fn decode_chunk_reads_every_record_batch_in_a_multi_batch_stream() {
864
+ use arrow::ipc::writer::StreamWriter;
865
+
866
+ let batch_a = make_batch(vec![1, 2], vec![1.0, 2.0]);
867
+ let batch_b = make_batch(vec![3, 4, 5], vec![3.0, 4.0, 5.0]);
868
+
869
+ let mut buf = Vec::new();
870
+ {
871
+ let mut writer = StreamWriter::try_new(&mut buf, &batch_a.schema()).unwrap();
872
+ writer.write(&batch_a).unwrap();
873
+ writer.write(&batch_b).unwrap();
874
+ writer.finish().unwrap();
875
+ }
876
+
877
+ let batches = decode_chunk(&Bytes::from(buf)).unwrap();
878
+ assert_eq!(
879
+ batches.len(),
880
+ 2,
881
+ "both record batches in the stream must be decoded, not just the first"
882
+ );
883
+ let total_rows: usize = batches.iter().map(|b| b.num_rows()).sum();
884
+ assert_eq!(total_rows, 5);
885
+ }
886
+
887
+ /// Regression test for a bug caught in code review before it shipped: an
888
+ /// empty (zero-byte) blob made `StreamDecoder`'s decode loop a no-op and
889
+ /// `decoder.finish()` saw a still-pristine state, which it treats as a
890
+ /// clean empty stream -- silently returning zero batches with no error at
891
+ /// all, instead of the loud failure the old `StreamReader`-based version
892
+ /// gave on the same input ("Expected schema message, found empty
893
+ /// stream"). Same silent-truncation shape as the real multi-frame LZ4 bug
894
+ /// this crate already shipped once (see `client.rs`'s
895
+ /// `decompress_lz4_frame` doc comment) -- a genuinely empty chunk blob
896
+ /// must never be mistaken for "legitimately zero rows".
897
+ #[test]
898
+ fn decode_chunk_rejects_an_empty_blob() {
899
+ let err = decode_chunk(&Bytes::new()).expect_err("an empty blob must error, not silently decode to zero rows");
900
+ assert!(
901
+ err.message.contains("empty"),
902
+ "error should mention the blob was empty: {}",
903
+ err.message
904
+ );
905
+ }
906
+
907
+ /// Companion to `decode_chunk_rejects_an_empty_blob` -- proves the empty-
908
+ /// blob check doesn't overcorrect: a *non-empty* stream containing only a
909
+ /// schema message and no `RecordBatch` at all (a legitimate shape for a
910
+ /// genuinely empty query result) must still decode successfully to zero
911
+ /// batches, not error.
912
+ #[test]
913
+ fn decode_chunk_accepts_a_schema_only_stream_with_zero_batches() {
914
+ use arrow::datatypes::{Field, Schema};
915
+ use arrow::ipc::writer::StreamWriter;
916
+
917
+ let schema = Arc::new(Schema::new(vec![Field::new("id", DataType::Int64, false)]));
918
+ let mut buf = Vec::new();
919
+ {
920
+ let mut writer = StreamWriter::try_new(&mut buf, &schema).unwrap();
921
+ writer.finish().unwrap();
922
+ }
923
+
924
+ let batches = decode_chunk(&Bytes::from(buf)).unwrap();
925
+ assert_eq!(
926
+ batches.len(),
927
+ 0,
928
+ "a schema-only stream with no batches is valid, not an error"
929
+ );
930
+ }
931
+
932
+ /// Regression/documentation test for a real behavior change found in code
933
+ /// review: `StreamDecoder` (unlike the old `StreamReader`) hard-errors on
934
+ /// any bytes left over after a stream's own EOS marker, instead of
935
+ /// silently ignoring them. Locking this in deliberately -- erroring beats
936
+ /// silently dropping whatever came after the truncation point, same
937
+ /// reasoning as the empty-blob check above -- even though real Databricks
938
+ /// chunks have not been observed to have trailing bytes.
939
+ #[test]
940
+ fn decode_chunk_errors_on_trailing_bytes_after_a_complete_stream() {
941
+ use arrow::ipc::writer::StreamWriter;
942
+
943
+ let batch = make_batch(vec![1, 2], vec![1.0, 2.0]);
944
+ let mut buf = Vec::new();
945
+ {
946
+ let mut writer = StreamWriter::try_new(&mut buf, &batch.schema()).unwrap();
947
+ writer.write(&batch).unwrap();
948
+ writer.finish().unwrap();
949
+ }
950
+ buf.extend_from_slice(&[0xAA; 8]);
951
+
952
+ decode_chunk(&Bytes::from(buf)).expect_err("trailing bytes after a complete stream's EOS marker must error");
953
+ }
954
+
955
+ /// Diagnostic only, not a correctness check (relative timing is too
956
+ /// flaky for CI) -- `cargo test --release -- --ignored --nocapture
957
+ /// decode_chunk_speed` to compare the current `StreamDecoder`-based
958
+ /// `decode_chunk` against the old `StreamReader`-based approach it
959
+ /// replaced, on a batch shaped like a real chunk (120 columns, mixed
960
+ /// types, 50k rows -- this session's own real-table benchmark).
961
+ #[test]
962
+ #[ignore]
963
+ fn decode_chunk_speed_vs_stream_reader() {
964
+ use arrow::array::{Float64Array, Int64Array, StringArray};
965
+ use arrow::datatypes::{Field, Schema};
966
+ use arrow::ipc::writer::StreamWriter;
967
+ use std::io::Cursor as IoCursor;
968
+ use std::time::Instant;
969
+
970
+ const ROWS: usize = 50_000;
971
+ const COLS: usize = 120;
972
+
973
+ let mut fields = Vec::with_capacity(COLS);
974
+ let mut columns: Vec<Arc<dyn Array>> = Vec::with_capacity(COLS);
975
+ for i in 0..COLS {
976
+ match i % 3 {
977
+ 0 => {
978
+ fields.push(Field::new(format!("c{i}"), DataType::Int64, false));
979
+ columns.push(Arc::new(Int64Array::from((0..ROWS as i64).collect::<Vec<_>>())));
980
+ }
981
+ 1 => {
982
+ fields.push(Field::new(format!("c{i}"), DataType::Float64, false));
983
+ columns.push(Arc::new(Float64Array::from(
984
+ (0..ROWS).map(|r| r as f64 * 1.5).collect::<Vec<_>>(),
985
+ )));
986
+ }
987
+ _ => {
988
+ fields.push(Field::new(format!("c{i}"), DataType::Utf8, false));
989
+ columns.push(Arc::new(StringArray::from(
990
+ (0..ROWS).map(|r| format!("row-{r}")).collect::<Vec<_>>(),
991
+ )));
992
+ }
993
+ }
994
+ }
995
+ let schema = Arc::new(Schema::new(fields));
996
+ let batch = RecordBatch::try_new(schema.clone(), columns).unwrap();
997
+
998
+ let mut buf = Vec::new();
999
+ {
1000
+ let mut writer = StreamWriter::try_new(&mut buf, &schema).unwrap();
1001
+ writer.write(&batch).unwrap();
1002
+ writer.finish().unwrap();
1003
+ }
1004
+ let blob = Bytes::from(buf);
1005
+
1006
+ const ITERS: u32 = 30;
1007
+
1008
+ // Old approach: StreamReader over an IoCursor -- copies every
1009
+ // column's data into freshly allocated buffers on every decode.
1010
+ let old_start = Instant::now();
1011
+ for _ in 0..ITERS {
1012
+ let reader = arrow::ipc::reader::StreamReader::try_new(IoCursor::new(&blob[..]), None).unwrap();
1013
+ let batches: Vec<RecordBatch> = reader.collect::<Result<Vec<_>, _>>().unwrap();
1014
+ assert_eq!(batches[0].num_rows(), ROWS);
1015
+ }
1016
+ let old_elapsed = old_start.elapsed();
1017
+
1018
+ // New approach: this file's actual decode_chunk.
1019
+ let new_start = Instant::now();
1020
+ for _ in 0..ITERS {
1021
+ let batches = decode_chunk(&blob).unwrap();
1022
+ assert_eq!(batches[0].num_rows(), ROWS);
1023
+ }
1024
+ let new_elapsed = new_start.elapsed();
1025
+
1026
+ println!(
1027
+ "decode_chunk speed ({COLS} cols x {ROWS} rows, {ITERS} iters): \
1028
+ StreamReader (old) = {old_elapsed:?} ({:?}/iter), \
1029
+ StreamDecoder (new) = {new_elapsed:?} ({:?}/iter)",
1030
+ old_elapsed / ITERS,
1031
+ new_elapsed / ITERS,
1032
+ );
1033
+ }
800
1034
  }
File without changes
File without changes