arrowbricks 1.3.1__tar.gz → 1.3.2__tar.gz

This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
Files changed (30) hide show
  1. {arrowbricks-1.3.1 → arrowbricks-1.3.2}/PKG-INFO +1 -1
  2. {arrowbricks-1.3.1 → arrowbricks-1.3.2}/pyproject.toml +1 -1
  3. {arrowbricks-1.3.1 → arrowbricks-1.3.2}/rust/arrowbricks_core/Cargo.lock +1 -1
  4. {arrowbricks-1.3.1 → arrowbricks-1.3.2}/rust/arrowbricks_core/Cargo.toml +1 -1
  5. {arrowbricks-1.3.1 → arrowbricks-1.3.2}/rust/arrowbricks_core/src/pipeline.rs +95 -2
  6. {arrowbricks-1.3.1 → arrowbricks-1.3.2}/LICENSE +0 -0
  7. {arrowbricks-1.3.1 → arrowbricks-1.3.2}/README.md +0 -0
  8. {arrowbricks-1.3.1 → arrowbricks-1.3.2}/rust/arrowbricks_core/.gitignore +0 -0
  9. {arrowbricks-1.3.1 → arrowbricks-1.3.2}/rust/arrowbricks_core/README.md +0 -0
  10. {arrowbricks-1.3.1 → arrowbricks-1.3.2}/rust/arrowbricks_core/examples/duckdb_query.py +0 -0
  11. {arrowbricks-1.3.1 → arrowbricks-1.3.2}/rust/arrowbricks_core/examples/fastapi_sse.py +0 -0
  12. {arrowbricks-1.3.1 → arrowbricks-1.3.2}/rust/arrowbricks_core/rustfmt.toml +0 -0
  13. {arrowbricks-1.3.1 → arrowbricks-1.3.2}/rust/arrowbricks_core/src/client.rs +0 -0
  14. {arrowbricks-1.3.1 → arrowbricks-1.3.2}/rust/arrowbricks_core/src/heartbeat.rs +0 -0
  15. {arrowbricks-1.3.1 → arrowbricks-1.3.2}/rust/arrowbricks_core/src/lib.rs +0 -0
  16. {arrowbricks-1.3.1 → arrowbricks-1.3.2}/rust/arrowbricks_core/tests/wiremock_pipeline.rs +0 -0
  17. {arrowbricks-1.3.1 → arrowbricks-1.3.2}/rust/arrowbricks_core/tests/wiremock_volume_files.rs +0 -0
  18. {arrowbricks-1.3.1 → arrowbricks-1.3.2}/rust/arrowbricks_core/tests_py/test_execute_json.py +0 -0
  19. {arrowbricks-1.3.1 → arrowbricks-1.3.2}/rust/arrowbricks_core/tests_py/test_ipc_stream.py +0 -0
  20. {arrowbricks-1.3.1 → arrowbricks-1.3.2}/rust/arrowbricks_core/tests_py/test_parameters.py +0 -0
  21. {arrowbricks-1.3.1 → arrowbricks-1.3.2}/rust/arrowbricks_core/tests_py/test_stream_ndjson_lines.py +0 -0
  22. {arrowbricks-1.3.1 → arrowbricks-1.3.2}/rust/arrowbricks_core/tests_py/test_streaming.py +0 -0
  23. {arrowbricks-1.3.1 → arrowbricks-1.3.2}/rust/arrowbricks_core/tests_py/test_token_provider.py +0 -0
  24. {arrowbricks-1.3.1 → arrowbricks-1.3.2}/rust/arrowbricks_core/tests_py/test_volume_files.py +0 -0
  25. {arrowbricks-1.3.1 → arrowbricks-1.3.2}/src/arrowbricks/__init__.py +0 -0
  26. {arrowbricks-1.3.1 → arrowbricks-1.3.2}/src/arrowbricks/_core.pyi +0 -0
  27. {arrowbricks-1.3.1 → arrowbricks-1.3.2}/src/arrowbricks/_streaming.py +0 -0
  28. {arrowbricks-1.3.1 → arrowbricks-1.3.2}/src/arrowbricks/client.py +0 -0
  29. {arrowbricks-1.3.1 → arrowbricks-1.3.2}/src/arrowbricks/cursor.py +0 -0
  30. {arrowbricks-1.3.1 → arrowbricks-1.3.2}/src/arrowbricks/py.typed +0 -0
@@ -1,6 +1,6 @@
1
1
  Metadata-Version: 2.4
2
2
  Name: arrowbricks
3
- Version: 1.3.1
3
+ Version: 1.3.2
4
4
  Requires-Dist: arro3-core>=0.8 ; extra == 'arro3'
5
5
  Provides-Extra: arro3
6
6
  License-File: LICENSE
@@ -1,6 +1,6 @@
1
1
  [project]
2
2
  name = "arrowbricks"
3
- version = "1.3.1"
3
+ version = "1.3.2"
4
4
  description = "Runs SQL against a Databricks SQL warehouse via the Statement Execution API and hands you the result as Arrow -- a DB-API-ish Cursor (fetchone/fetchmany/fetchall/fetchall_arrow) or NDJSON streaming. Rust/PyO3 core throughout -- zero required runtime dependencies."
5
5
  readme = "README.md"
6
6
  license = "MIT"
@@ -240,7 +240,7 @@ dependencies = [
240
240
 
241
241
  [[package]]
242
242
  name = "arrowbricks_core"
243
- version = "1.3.1"
243
+ version = "1.3.2"
244
244
  dependencies = [
245
245
  "arrow",
246
246
  "arrow-json",
@@ -1,6 +1,6 @@
1
1
  [package]
2
2
  name = "arrowbricks_core"
3
- version = "1.3.1"
3
+ version = "1.3.2"
4
4
  edition = "2024"
5
5
  readme = "README.md"
6
6
 
@@ -109,9 +109,34 @@ impl ExecuteResult {
109
109
  /// writes it, not this crate), decoded batches now slice directly into the
110
110
  /// same allocation `blob` already held since the network fetch, cutting out
111
111
  /// a second full copy of every chunk's bytes; `require_alignment` stays at
112
- /// its default `false`, so a misaligned buffer still falls back to a copy
113
- /// automatically rather than erroring (arrow-ipc's own documented behavior).
112
+ /// its default `false`, so a misaligned *fixed-width* buffer still falls back
113
+ /// to a copy automatically rather than erroring (arrow-ipc's own documented
114
+ /// behavior) -- variable-width values, null bitmaps, and nested/dictionary
115
+ /// children all stay zero-copy regardless. One exception this doesn't cover:
116
+ /// if a `RecordBatch` message declared IPC *buffer*-level compression (a
117
+ /// different, unrelated feature from this crate's own cloud-fetch
118
+ /// `result_compression` unwrap in `client.rs`, which already ran before this
119
+ /// function ever sees the bytes), `arrow-ipc`'s own reader always
120
+ /// decompresses into fresh buffers there -- not something Databricks has
121
+ /// been observed to use in this format, but not something this crate
122
+ /// controls either.
123
+ ///
124
+ /// A zero-length `blob` is rejected explicitly rather than handed to
125
+ /// `StreamDecoder`: found in code review that an empty buffer makes the
126
+ /// `while` loop below a no-op and `decoder.finish()` sees a still-pristine
127
+ /// decoder state, which its own `Ok(())` arm treats as a *clean, empty*
128
+ /// stream -- silently returning zero batches with no error at all, the same
129
+ /// silent-truncation failure mode as the real multi-frame LZ4 bug this crate
130
+ /// already shipped once (see the `result_compression` invariant above). The
131
+ /// old `StreamReader`-based version failed loudly on this input instead
132
+ /// ("Expected schema message, found empty stream"); this restores that.
114
133
  fn decode_chunk(blob: &Bytes) -> Result<Vec<RecordBatch>, ApiError> {
134
+ if blob.is_empty() {
135
+ return Err(ApiError {
136
+ message: "empty Arrow IPC chunk: expected at least a schema message".to_string(),
137
+ transient: false,
138
+ });
139
+ }
115
140
  let mut buffer = ArrowBuffer::from(blob.clone());
116
141
  let mut decoder = StreamDecoder::new();
117
142
  let mut batches = Vec::new();
@@ -859,6 +884,74 @@ mod tests {
859
884
  assert_eq!(total_rows, 5);
860
885
  }
861
886
 
887
+ /// Regression test for a bug caught in code review before it shipped: an
888
+ /// empty (zero-byte) blob made `StreamDecoder`'s decode loop a no-op and
889
+ /// `decoder.finish()` saw a still-pristine state, which it treats as a
890
+ /// clean empty stream -- silently returning zero batches with no error at
891
+ /// all, instead of the loud failure the old `StreamReader`-based version
892
+ /// gave on the same input ("Expected schema message, found empty
893
+ /// stream"). Same silent-truncation shape as the real multi-frame LZ4 bug
894
+ /// this crate already shipped once (see `client.rs`'s
895
+ /// `decompress_lz4_frame` doc comment) -- a genuinely empty chunk blob
896
+ /// must never be mistaken for "legitimately zero rows".
897
+ #[test]
898
+ fn decode_chunk_rejects_an_empty_blob() {
899
+ let err = decode_chunk(&Bytes::new()).expect_err("an empty blob must error, not silently decode to zero rows");
900
+ assert!(
901
+ err.message.contains("empty"),
902
+ "error should mention the blob was empty: {}",
903
+ err.message
904
+ );
905
+ }
906
+
907
+ /// Companion to `decode_chunk_rejects_an_empty_blob` -- proves the empty-
908
+ /// blob check doesn't overcorrect: a *non-empty* stream containing only a
909
+ /// schema message and no `RecordBatch` at all (a legitimate shape for a
910
+ /// genuinely empty query result) must still decode successfully to zero
911
+ /// batches, not error.
912
+ #[test]
913
+ fn decode_chunk_accepts_a_schema_only_stream_with_zero_batches() {
914
+ use arrow::datatypes::{Field, Schema};
915
+ use arrow::ipc::writer::StreamWriter;
916
+
917
+ let schema = Arc::new(Schema::new(vec![Field::new("id", DataType::Int64, false)]));
918
+ let mut buf = Vec::new();
919
+ {
920
+ let mut writer = StreamWriter::try_new(&mut buf, &schema).unwrap();
921
+ writer.finish().unwrap();
922
+ }
923
+
924
+ let batches = decode_chunk(&Bytes::from(buf)).unwrap();
925
+ assert_eq!(
926
+ batches.len(),
927
+ 0,
928
+ "a schema-only stream with no batches is valid, not an error"
929
+ );
930
+ }
931
+
932
+ /// Regression/documentation test for a real behavior change found in code
933
+ /// review: `StreamDecoder` (unlike the old `StreamReader`) hard-errors on
934
+ /// any bytes left over after a stream's own EOS marker, instead of
935
+ /// silently ignoring them. Locking this in deliberately -- erroring beats
936
+ /// silently dropping whatever came after the truncation point, same
937
+ /// reasoning as the empty-blob check above -- even though real Databricks
938
+ /// chunks have not been observed to have trailing bytes.
939
+ #[test]
940
+ fn decode_chunk_errors_on_trailing_bytes_after_a_complete_stream() {
941
+ use arrow::ipc::writer::StreamWriter;
942
+
943
+ let batch = make_batch(vec![1, 2], vec![1.0, 2.0]);
944
+ let mut buf = Vec::new();
945
+ {
946
+ let mut writer = StreamWriter::try_new(&mut buf, &batch.schema()).unwrap();
947
+ writer.write(&batch).unwrap();
948
+ writer.finish().unwrap();
949
+ }
950
+ buf.extend_from_slice(&[0xAA; 8]);
951
+
952
+ decode_chunk(&Bytes::from(buf)).expect_err("trailing bytes after a complete stream's EOS marker must error");
953
+ }
954
+
862
955
  /// Diagnostic only, not a correctness check (relative timing is too
863
956
  /// flaky for CI) -- `cargo test --release -- --ignored --nocapture
864
957
  /// decode_chunk_speed` to compare the current `StreamDecoder`-based
File without changes
File without changes