arrowbricks 1.3.1__tar.gz → 1.3.2__tar.gz
This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
- {arrowbricks-1.3.1 → arrowbricks-1.3.2}/PKG-INFO +1 -1
- {arrowbricks-1.3.1 → arrowbricks-1.3.2}/pyproject.toml +1 -1
- {arrowbricks-1.3.1 → arrowbricks-1.3.2}/rust/arrowbricks_core/Cargo.lock +1 -1
- {arrowbricks-1.3.1 → arrowbricks-1.3.2}/rust/arrowbricks_core/Cargo.toml +1 -1
- {arrowbricks-1.3.1 → arrowbricks-1.3.2}/rust/arrowbricks_core/src/pipeline.rs +95 -2
- {arrowbricks-1.3.1 → arrowbricks-1.3.2}/LICENSE +0 -0
- {arrowbricks-1.3.1 → arrowbricks-1.3.2}/README.md +0 -0
- {arrowbricks-1.3.1 → arrowbricks-1.3.2}/rust/arrowbricks_core/.gitignore +0 -0
- {arrowbricks-1.3.1 → arrowbricks-1.3.2}/rust/arrowbricks_core/README.md +0 -0
- {arrowbricks-1.3.1 → arrowbricks-1.3.2}/rust/arrowbricks_core/examples/duckdb_query.py +0 -0
- {arrowbricks-1.3.1 → arrowbricks-1.3.2}/rust/arrowbricks_core/examples/fastapi_sse.py +0 -0
- {arrowbricks-1.3.1 → arrowbricks-1.3.2}/rust/arrowbricks_core/rustfmt.toml +0 -0
- {arrowbricks-1.3.1 → arrowbricks-1.3.2}/rust/arrowbricks_core/src/client.rs +0 -0
- {arrowbricks-1.3.1 → arrowbricks-1.3.2}/rust/arrowbricks_core/src/heartbeat.rs +0 -0
- {arrowbricks-1.3.1 → arrowbricks-1.3.2}/rust/arrowbricks_core/src/lib.rs +0 -0
- {arrowbricks-1.3.1 → arrowbricks-1.3.2}/rust/arrowbricks_core/tests/wiremock_pipeline.rs +0 -0
- {arrowbricks-1.3.1 → arrowbricks-1.3.2}/rust/arrowbricks_core/tests/wiremock_volume_files.rs +0 -0
- {arrowbricks-1.3.1 → arrowbricks-1.3.2}/rust/arrowbricks_core/tests_py/test_execute_json.py +0 -0
- {arrowbricks-1.3.1 → arrowbricks-1.3.2}/rust/arrowbricks_core/tests_py/test_ipc_stream.py +0 -0
- {arrowbricks-1.3.1 → arrowbricks-1.3.2}/rust/arrowbricks_core/tests_py/test_parameters.py +0 -0
- {arrowbricks-1.3.1 → arrowbricks-1.3.2}/rust/arrowbricks_core/tests_py/test_stream_ndjson_lines.py +0 -0
- {arrowbricks-1.3.1 → arrowbricks-1.3.2}/rust/arrowbricks_core/tests_py/test_streaming.py +0 -0
- {arrowbricks-1.3.1 → arrowbricks-1.3.2}/rust/arrowbricks_core/tests_py/test_token_provider.py +0 -0
- {arrowbricks-1.3.1 → arrowbricks-1.3.2}/rust/arrowbricks_core/tests_py/test_volume_files.py +0 -0
- {arrowbricks-1.3.1 → arrowbricks-1.3.2}/src/arrowbricks/__init__.py +0 -0
- {arrowbricks-1.3.1 → arrowbricks-1.3.2}/src/arrowbricks/_core.pyi +0 -0
- {arrowbricks-1.3.1 → arrowbricks-1.3.2}/src/arrowbricks/_streaming.py +0 -0
- {arrowbricks-1.3.1 → arrowbricks-1.3.2}/src/arrowbricks/client.py +0 -0
- {arrowbricks-1.3.1 → arrowbricks-1.3.2}/src/arrowbricks/cursor.py +0 -0
- {arrowbricks-1.3.1 → arrowbricks-1.3.2}/src/arrowbricks/py.typed +0 -0
|
@@ -1,6 +1,6 @@
|
|
|
1
1
|
[project]
|
|
2
2
|
name = "arrowbricks"
|
|
3
|
-
version = "1.3.
|
|
3
|
+
version = "1.3.2"
|
|
4
4
|
description = "Runs SQL against a Databricks SQL warehouse via the Statement Execution API and hands you the result as Arrow -- a DB-API-ish Cursor (fetchone/fetchmany/fetchall/fetchall_arrow) or NDJSON streaming. Rust/PyO3 core throughout -- zero required runtime dependencies."
|
|
5
5
|
readme = "README.md"
|
|
6
6
|
license = "MIT"
|
|
@@ -109,9 +109,34 @@ impl ExecuteResult {
|
|
|
109
109
|
/// writes it, not this crate), decoded batches now slice directly into the
|
|
110
110
|
/// same allocation `blob` already held since the network fetch, cutting out
|
|
111
111
|
/// a second full copy of every chunk's bytes; `require_alignment` stays at
|
|
112
|
-
/// its default `false`, so a misaligned buffer still falls back
|
|
113
|
-
/// automatically rather than erroring (arrow-ipc's own documented
|
|
112
|
+
/// its default `false`, so a misaligned *fixed-width* buffer still falls back
|
|
113
|
+
/// to a copy automatically rather than erroring (arrow-ipc's own documented
|
|
114
|
+
/// behavior) -- variable-width values, null bitmaps, and nested/dictionary
|
|
115
|
+
/// children all stay zero-copy regardless. One exception this doesn't cover:
|
|
116
|
+
/// if a `RecordBatch` message declared IPC *buffer*-level compression (a
|
|
117
|
+
/// different, unrelated feature from this crate's own cloud-fetch
|
|
118
|
+
/// `result_compression` unwrap in `client.rs`, which already ran before this
|
|
119
|
+
/// function ever sees the bytes), `arrow-ipc`'s own reader always
|
|
120
|
+
/// decompresses into fresh buffers there -- not something Databricks has
|
|
121
|
+
/// been observed to use in this format, but not something this crate
|
|
122
|
+
/// controls either.
|
|
123
|
+
///
|
|
124
|
+
/// A zero-length `blob` is rejected explicitly rather than handed to
|
|
125
|
+
/// `StreamDecoder`: found in code review that an empty buffer makes the
|
|
126
|
+
/// `while` loop below a no-op and `decoder.finish()` sees a still-pristine
|
|
127
|
+
/// decoder state, which its own `Ok(())` arm treats as a *clean, empty*
|
|
128
|
+
/// stream -- silently returning zero batches with no error at all, the same
|
|
129
|
+
/// silent-truncation failure mode as the real multi-frame LZ4 bug this crate
|
|
130
|
+
/// already shipped once (see the `result_compression` invariant above). The
|
|
131
|
+
/// old `StreamReader`-based version failed loudly on this input instead
|
|
132
|
+
/// ("Expected schema message, found empty stream"); this restores that.
|
|
114
133
|
fn decode_chunk(blob: &Bytes) -> Result<Vec<RecordBatch>, ApiError> {
|
|
134
|
+
if blob.is_empty() {
|
|
135
|
+
return Err(ApiError {
|
|
136
|
+
message: "empty Arrow IPC chunk: expected at least a schema message".to_string(),
|
|
137
|
+
transient: false,
|
|
138
|
+
});
|
|
139
|
+
}
|
|
115
140
|
let mut buffer = ArrowBuffer::from(blob.clone());
|
|
116
141
|
let mut decoder = StreamDecoder::new();
|
|
117
142
|
let mut batches = Vec::new();
|
|
@@ -859,6 +884,74 @@ mod tests {
|
|
|
859
884
|
assert_eq!(total_rows, 5);
|
|
860
885
|
}
|
|
861
886
|
|
|
887
|
+
/// Regression test for a bug caught in code review before it shipped: an
|
|
888
|
+
/// empty (zero-byte) blob made `StreamDecoder`'s decode loop a no-op and
|
|
889
|
+
/// `decoder.finish()` saw a still-pristine state, which it treats as a
|
|
890
|
+
/// clean empty stream -- silently returning zero batches with no error at
|
|
891
|
+
/// all, instead of the loud failure the old `StreamReader`-based version
|
|
892
|
+
/// gave on the same input ("Expected schema message, found empty
|
|
893
|
+
/// stream"). Same silent-truncation shape as the real multi-frame LZ4 bug
|
|
894
|
+
/// this crate already shipped once (see `client.rs`'s
|
|
895
|
+
/// `decompress_lz4_frame` doc comment) -- a genuinely empty chunk blob
|
|
896
|
+
/// must never be mistaken for "legitimately zero rows".
|
|
897
|
+
#[test]
|
|
898
|
+
fn decode_chunk_rejects_an_empty_blob() {
|
|
899
|
+
let err = decode_chunk(&Bytes::new()).expect_err("an empty blob must error, not silently decode to zero rows");
|
|
900
|
+
assert!(
|
|
901
|
+
err.message.contains("empty"),
|
|
902
|
+
"error should mention the blob was empty: {}",
|
|
903
|
+
err.message
|
|
904
|
+
);
|
|
905
|
+
}
|
|
906
|
+
|
|
907
|
+
/// Companion to `decode_chunk_rejects_an_empty_blob` -- proves the empty-
|
|
908
|
+
/// blob check doesn't overcorrect: a *non-empty* stream containing only a
|
|
909
|
+
/// schema message and no `RecordBatch` at all (a legitimate shape for a
|
|
910
|
+
/// genuinely empty query result) must still decode successfully to zero
|
|
911
|
+
/// batches, not error.
|
|
912
|
+
#[test]
|
|
913
|
+
fn decode_chunk_accepts_a_schema_only_stream_with_zero_batches() {
|
|
914
|
+
use arrow::datatypes::{Field, Schema};
|
|
915
|
+
use arrow::ipc::writer::StreamWriter;
|
|
916
|
+
|
|
917
|
+
let schema = Arc::new(Schema::new(vec![Field::new("id", DataType::Int64, false)]));
|
|
918
|
+
let mut buf = Vec::new();
|
|
919
|
+
{
|
|
920
|
+
let mut writer = StreamWriter::try_new(&mut buf, &schema).unwrap();
|
|
921
|
+
writer.finish().unwrap();
|
|
922
|
+
}
|
|
923
|
+
|
|
924
|
+
let batches = decode_chunk(&Bytes::from(buf)).unwrap();
|
|
925
|
+
assert_eq!(
|
|
926
|
+
batches.len(),
|
|
927
|
+
0,
|
|
928
|
+
"a schema-only stream with no batches is valid, not an error"
|
|
929
|
+
);
|
|
930
|
+
}
|
|
931
|
+
|
|
932
|
+
/// Regression/documentation test for a real behavior change found in code
|
|
933
|
+
/// review: `StreamDecoder` (unlike the old `StreamReader`) hard-errors on
|
|
934
|
+
/// any bytes left over after a stream's own EOS marker, instead of
|
|
935
|
+
/// silently ignoring them. Locking this in deliberately -- erroring beats
|
|
936
|
+
/// silently dropping whatever came after the truncation point, same
|
|
937
|
+
/// reasoning as the empty-blob check above -- even though real Databricks
|
|
938
|
+
/// chunks have not been observed to have trailing bytes.
|
|
939
|
+
#[test]
|
|
940
|
+
fn decode_chunk_errors_on_trailing_bytes_after_a_complete_stream() {
|
|
941
|
+
use arrow::ipc::writer::StreamWriter;
|
|
942
|
+
|
|
943
|
+
let batch = make_batch(vec![1, 2], vec![1.0, 2.0]);
|
|
944
|
+
let mut buf = Vec::new();
|
|
945
|
+
{
|
|
946
|
+
let mut writer = StreamWriter::try_new(&mut buf, &batch.schema()).unwrap();
|
|
947
|
+
writer.write(&batch).unwrap();
|
|
948
|
+
writer.finish().unwrap();
|
|
949
|
+
}
|
|
950
|
+
buf.extend_from_slice(&[0xAA; 8]);
|
|
951
|
+
|
|
952
|
+
decode_chunk(&Bytes::from(buf)).expect_err("trailing bytes after a complete stream's EOS marker must error");
|
|
953
|
+
}
|
|
954
|
+
|
|
862
955
|
/// Diagnostic only, not a correctness check (relative timing is too
|
|
863
956
|
/// flaky for CI) -- `cargo test --release -- --ignored --nocapture
|
|
864
957
|
/// decode_chunk_speed` to compare the current `StreamDecoder`-based
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
{arrowbricks-1.3.1 → arrowbricks-1.3.2}/rust/arrowbricks_core/tests/wiremock_volume_files.rs
RENAMED
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
{arrowbricks-1.3.1 → arrowbricks-1.3.2}/rust/arrowbricks_core/tests_py/test_stream_ndjson_lines.py
RENAMED
|
File without changes
|
|
File without changes
|
{arrowbricks-1.3.1 → arrowbricks-1.3.2}/rust/arrowbricks_core/tests_py/test_token_provider.py
RENAMED
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|