arrowbricks 3.1.0__tar.gz → 3.1.2__tar.gz
This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
- {arrowbricks-3.1.0 → arrowbricks-3.1.2}/PKG-INFO +1 -1
- {arrowbricks-3.1.0 → arrowbricks-3.1.2}/pyproject.toml +1 -1
- {arrowbricks-3.1.0 → arrowbricks-3.1.2}/rust/arrowbricks_core/Cargo.lock +2 -1
- {arrowbricks-3.1.0 → arrowbricks-3.1.2}/rust/arrowbricks_core/Cargo.toml +7 -1
- arrowbricks-3.1.2/rust/arrowbricks_core/src/client/download.rs +496 -0
- arrowbricks-3.1.2/rust/arrowbricks_core/src/client/error.rs +167 -0
- arrowbricks-3.1.2/rust/arrowbricks_core/src/client/model.rs +284 -0
- arrowbricks-3.1.2/rust/arrowbricks_core/src/client/sea.rs +670 -0
- arrowbricks-3.1.2/rust/arrowbricks_core/src/client/thrift_rpc.rs +208 -0
- arrowbricks-3.1.2/rust/arrowbricks_core/src/client/volume.rs +69 -0
- arrowbricks-3.1.2/rust/arrowbricks_core/src/client.rs +757 -0
- {arrowbricks-3.1.0 → arrowbricks-3.1.2}/rust/arrowbricks_core/src/heartbeat.rs +112 -107
- {arrowbricks-3.1.0 → arrowbricks-3.1.2}/rust/arrowbricks_core/src/json_convert.rs +11 -35
- arrowbricks-3.1.2/rust/arrowbricks_core/src/pipeline/ndjson.rs +457 -0
- arrowbricks-3.1.2/rust/arrowbricks_core/src/pipeline/reorder.rs +623 -0
- arrowbricks-3.1.2/rust/arrowbricks_core/src/pipeline/sea.rs +521 -0
- arrowbricks-3.1.2/rust/arrowbricks_core/src/pipeline/stats.rs +299 -0
- arrowbricks-3.1.2/rust/arrowbricks_core/src/pipeline/test_support.rs +29 -0
- arrowbricks-3.1.2/rust/arrowbricks_core/src/pipeline/thrift_exec.rs +845 -0
- arrowbricks-3.1.2/rust/arrowbricks_core/src/pipeline.rs +32 -0
- arrowbricks-3.1.0/rust/arrowbricks_core/src/client.rs +0 -2492
- arrowbricks-3.1.0/rust/arrowbricks_core/src/pipeline.rs +0 -2652
- {arrowbricks-3.1.0 → arrowbricks-3.1.2}/LICENSE +0 -0
- {arrowbricks-3.1.0 → arrowbricks-3.1.2}/README.md +0 -0
- {arrowbricks-3.1.0 → arrowbricks-3.1.2}/rust/arrowbricks_core/.gitignore +0 -0
- {arrowbricks-3.1.0 → arrowbricks-3.1.2}/rust/arrowbricks_core/README.md +0 -0
- {arrowbricks-3.1.0 → arrowbricks-3.1.2}/rust/arrowbricks_core/examples/duckdb_query.py +0 -0
- {arrowbricks-3.1.0 → arrowbricks-3.1.2}/rust/arrowbricks_core/examples/fastapi_sse.py +0 -0
- {arrowbricks-3.1.0 → arrowbricks-3.1.2}/rust/arrowbricks_core/rustfmt.toml +0 -0
- {arrowbricks-3.1.0 → arrowbricks-3.1.2}/rust/arrowbricks_core/src/lib.rs +0 -0
- {arrowbricks-3.1.0 → arrowbricks-3.1.2}/rust/arrowbricks_core/src/thrift.rs +0 -0
- {arrowbricks-3.1.0 → arrowbricks-3.1.2}/rust/arrowbricks_core/tests/common/mod.rs +0 -0
- {arrowbricks-3.1.0 → arrowbricks-3.1.2}/rust/arrowbricks_core/tests/wiremock_pipeline.rs +0 -0
- {arrowbricks-3.1.0 → arrowbricks-3.1.2}/rust/arrowbricks_core/tests/wiremock_thrift.rs +0 -0
- {arrowbricks-3.1.0 → arrowbricks-3.1.2}/rust/arrowbricks_core/tests/wiremock_volume_files.rs +0 -0
- {arrowbricks-3.1.0 → arrowbricks-3.1.2}/rust/arrowbricks_core/tests_py/conftest.py +0 -0
- {arrowbricks-3.1.0 → arrowbricks-3.1.2}/rust/arrowbricks_core/tests_py/test_ipc_stream.py +0 -0
- {arrowbricks-3.1.0 → arrowbricks-3.1.2}/rust/arrowbricks_core/tests_py/test_parameters.py +0 -0
- {arrowbricks-3.1.0 → arrowbricks-3.1.2}/rust/arrowbricks_core/tests_py/test_stream_ndjson_lines.py +0 -0
- {arrowbricks-3.1.0 → arrowbricks-3.1.2}/rust/arrowbricks_core/tests_py/test_streaming.py +0 -0
- {arrowbricks-3.1.0 → arrowbricks-3.1.2}/rust/arrowbricks_core/tests_py/test_thrift.py +0 -0
- {arrowbricks-3.1.0 → arrowbricks-3.1.2}/rust/arrowbricks_core/tests_py/test_token_provider.py +0 -0
- {arrowbricks-3.1.0 → arrowbricks-3.1.2}/rust/arrowbricks_core/tests_py/test_volume_files.py +0 -0
- {arrowbricks-3.1.0 → arrowbricks-3.1.2}/rust/arrowbricks_core/tests_py/thrift_mock.py +0 -0
- {arrowbricks-3.1.0 → arrowbricks-3.1.2}/src/arrowbricks/__init__.py +0 -0
- {arrowbricks-3.1.0 → arrowbricks-3.1.2}/src/arrowbricks/_core.pyi +0 -0
- {arrowbricks-3.1.0 → arrowbricks-3.1.2}/src/arrowbricks/_streaming.py +0 -0
- {arrowbricks-3.1.0 → arrowbricks-3.1.2}/src/arrowbricks/client.py +0 -0
- {arrowbricks-3.1.0 → arrowbricks-3.1.2}/src/arrowbricks/cursor.py +0 -0
- {arrowbricks-3.1.0 → arrowbricks-3.1.2}/src/arrowbricks/py.typed +0 -0
|
@@ -1,6 +1,6 @@
|
|
|
1
1
|
[project]
|
|
2
2
|
name = "arrowbricks"
|
|
3
|
-
version = "3.1.
|
|
3
|
+
version = "3.1.2"
|
|
4
4
|
description = "Runs SQL against a Databricks SQL warehouse via the Statement Execution API and hands you the result as Arrow -- a DB-API-ish Cursor (fetchone/fetchmany/fetchall/fetchall_arrow) or NDJSON streaming. Rust/PyO3 core throughout -- zero required runtime dependencies."
|
|
5
5
|
readme = "README.md"
|
|
6
6
|
license = "MIT"
|
|
@@ -1,6 +1,6 @@
|
|
|
1
1
|
[package]
|
|
2
2
|
name = "arrowbricks_core"
|
|
3
|
-
version = "3.1.
|
|
3
|
+
version = "3.1.2"
|
|
4
4
|
edition = "2024"
|
|
5
5
|
readme = "README.md"
|
|
6
6
|
|
|
@@ -15,6 +15,12 @@ crate-type = ["cdylib", "rlib"]
|
|
|
15
15
|
arrow = { version = "59.1.0", default-features = false, features = ["ipc"] }
|
|
16
16
|
arrow-json = "59.1.0"
|
|
17
17
|
bytes = "1.12.1"
|
|
18
|
+
# Already an unconditional transitive dependency of arrow-cast (pulled in by
|
|
19
|
+
# pyo3-arrow) at this exact version -- declared directly here too so
|
|
20
|
+
# json_convert.rs can decode BINARY columns without hand-rolling base64. No
|
|
21
|
+
# new code ships that wasn't already linked in (feature unification, same
|
|
22
|
+
# reasoning as chrono/rustls below).
|
|
23
|
+
base64 = "0.23"
|
|
18
24
|
# Already an unconditional transitive dependency of pyo3-arrow (see its own
|
|
19
25
|
# comment below) -- declared directly here too so json_convert.rs can parse
|
|
20
26
|
# DATE/TIMESTAMP/TIMESTAMP_NTZ strings without hand-rolling calendar
|
|
@@ -0,0 +1,496 @@
|
|
|
1
|
+
//! Downloading and decompressing one cloud-fetch external link: a retryable
|
|
2
|
+
//! GET (`fetch_link_bytes`), an adaptive-concurrency Range-split variant for
|
|
3
|
+
//! when the shared download-slot budget has room to spare
|
|
4
|
+
//! (`fetch_link_bytes_budgeted`/`fetch_link_bytes_split`), and unwrapping the
|
|
5
|
+
//! outer LZ4 Frame compression the server applies when `result_compression`
|
|
6
|
+
//! was requested (`decompress_lz4_frame`) -- shared by both the SEA and
|
|
7
|
+
//! Thrift backends' own chunk-fetch workers.
|
|
8
|
+
|
|
9
|
+
use std::sync::Arc;
|
|
10
|
+
use std::sync::atomic::Ordering;
|
|
11
|
+
|
|
12
|
+
use bytes::Bytes;
|
|
13
|
+
|
|
14
|
+
use super::DbClient;
|
|
15
|
+
use super::MAX_SPLIT_PARTS;
|
|
16
|
+
use super::error::{ApiError, join_error};
|
|
17
|
+
use super::model::QueryStatsAccumulator;
|
|
18
|
+
|
|
19
|
+
/// Unwraps one downloaded chunk file's outer LZ4 Frame compression --
|
|
20
|
+
/// server-side, this is a whole-file wrapper applied because we asked for
|
|
21
|
+
/// `result_compression: "LZ4_FRAME"`, not Arrow-IPC's own (unrelated,
|
|
22
|
+
/// per-buffer) compression option. `reqwest::Response::bytes()` already
|
|
23
|
+
/// collected the full body reliably before this runs, so a decode failure
|
|
24
|
+
/// here means a genuine format problem, not a network blip -- treated as
|
|
25
|
+
/// permanent (not retried), same reasoning as `ApiError::from_reqwest`.
|
|
26
|
+
///
|
|
27
|
+
/// A real Databricks chunk is not one LZ4 frame -- it's several concatenated
|
|
28
|
+
/// back to back (confirmed against a real workspace: a 50k-row/1MB chunk came
|
|
29
|
+
/// back as 18 separate frames). `FrameDecoder::read_to_end` stops the moment
|
|
30
|
+
/// it hits the first frame's own end marker; calling it once silently
|
|
31
|
+
/// produced only that first frame's ~200 bytes (just the Arrow schema
|
|
32
|
+
/// message, no data) instead of the full stream, which `pipeline/reorder.rs`'s
|
|
33
|
+
/// `decode_chunk` then decoded as a zero-row/zero-column result with no error at all -- this bug
|
|
34
|
+
/// shipped from a design that was only ever verified against synthetic
|
|
35
|
+
/// single-frame test data. One `FrameDecoder` resets its own frame state
|
|
36
|
+
/// after each `EndMark` and picks up the next concatenated frame on a
|
|
37
|
+
/// subsequent `read_to_end` call against the *same* instance (verified: the
|
|
38
|
+
/// decoder's position in the underlying byte slice carries over across
|
|
39
|
+
/// calls) -- so looping `read_to_end` on one decoder until it stops growing
|
|
40
|
+
/// `out` reads every frame without reconstructing a decoder per frame.
|
|
41
|
+
pub(crate) fn decompress_lz4_frame(compressed: &Bytes) -> Result<Bytes, ApiError> {
|
|
42
|
+
use std::io::Read;
|
|
43
|
+
// `compressed.len()` is a real lower bound, but LZ4 on Arrow-IPC data
|
|
44
|
+
// (long dictionary/offset-buffer runs, mostly-repeated bytes) typically
|
|
45
|
+
// compresses several-fold -- estimating just the lower bound means the
|
|
46
|
+
// real decompressed size almost always blows past initial capacity,
|
|
47
|
+
// paying for repeated doubling-and-copy growth on every chunk. `* 4` is
|
|
48
|
+
// a heuristic, not a guarantee (`Vec` still grows normally if it's wrong
|
|
49
|
+
// either way) -- just a better starting point than the guaranteed-too-
|
|
50
|
+
// small lower bound.
|
|
51
|
+
let mut out = Vec::with_capacity(compressed.len() * 4);
|
|
52
|
+
let mut decoder = lz4_flex::frame::FrameDecoder::new(&compressed[..]);
|
|
53
|
+
// Terminate on the *reader* being exhausted, not on "output stopped
|
|
54
|
+
// growing" -- found in code review that a frame which happens to decode
|
|
55
|
+
// to zero bytes (a real, valid LZ4 Frame shape: header + immediate
|
|
56
|
+
// EndMark) makes one `read_to_end` call return `Ok(0)` for that frame
|
|
57
|
+
// without erroring and without necessarily advancing into the next
|
|
58
|
+
// frame yet, which the old `out.len() == before` check read as "no more
|
|
59
|
+
// frames" -- silently dropping every subsequent concatenated frame with
|
|
60
|
+
// no error at all. Same silent-truncation shape as the original
|
|
61
|
+
// multi-frame bug this loop exists to fix. Looping while the reader
|
|
62
|
+
// still has bytes left (regardless of whether the last call grew `out`)
|
|
63
|
+
// is the correct fix -- verified against a zero-content frame sandwiched
|
|
64
|
+
// between two real ones, see `decompress_lz4_frame_survives_a_zero_content_frame_in_the_middle`.
|
|
65
|
+
while !decoder.get_ref().is_empty() {
|
|
66
|
+
decoder
|
|
67
|
+
.read_to_end(&mut out)
|
|
68
|
+
.map_err(|e| ApiError::permanent(format!("LZ4 frame decompress failed: {e}")))?;
|
|
69
|
+
}
|
|
70
|
+
Ok(Bytes::from(out))
|
|
71
|
+
}
|
|
72
|
+
|
|
73
|
+
impl DbClient {
|
|
74
|
+
/// Unauthenticated -- external links are presigned blob-storage URLs,
|
|
75
|
+
/// same as `_fetch_link_bytes` in the Python client. `compressed` decodes
|
|
76
|
+
/// the server's LZ4 Frame wrapper (see `client/sea.rs`'s `execute_statement`'s
|
|
77
|
+
/// `result_compression` request) before handing bytes onward -- the
|
|
78
|
+
/// downloaded file is smaller over the wire, but its content is opaque
|
|
79
|
+
/// (Arrow-IPC or JSON, depending on `format`) until unwrapped here.
|
|
80
|
+
/// Decompression runs on `spawn_blocking`, same reasoning as
|
|
81
|
+
/// `pipeline/reorder.rs`'s Arrow-IPC decode: it's real CPU work (multiple LZ4
|
|
82
|
+
/// frames per chunk, see `decompress_lz4_frame`), and running it inline
|
|
83
|
+
/// would block this task's tokio worker thread from polling anything
|
|
84
|
+
/// else scheduled on it -- other concurrent chunk fetches, heartbeat
|
|
85
|
+
/// timers -- for however long that takes.
|
|
86
|
+
pub(crate) async fn fetch_link_bytes(
|
|
87
|
+
&self,
|
|
88
|
+
url: &str,
|
|
89
|
+
compressed: bool,
|
|
90
|
+
stats: &QueryStatsAccumulator,
|
|
91
|
+
) -> Result<Bytes, ApiError> {
|
|
92
|
+
self.retry_call_tracked(Some(stats), || async {
|
|
93
|
+
let resp = self
|
|
94
|
+
.http
|
|
95
|
+
.get(url)
|
|
96
|
+
.timeout(self.http_timeout)
|
|
97
|
+
.send()
|
|
98
|
+
.await
|
|
99
|
+
.map_err(|e| ApiError::from_reqwest(e, true))?;
|
|
100
|
+
let status = resp.status();
|
|
101
|
+
if !status.is_success() {
|
|
102
|
+
let text = resp.text().await.unwrap_or_default();
|
|
103
|
+
return Err(ApiError::from_status(status, &text, true));
|
|
104
|
+
}
|
|
105
|
+
let bytes = resp.bytes().await.map_err(|e| ApiError::from_reqwest(e, true))?;
|
|
106
|
+
// Counted here, before decompression -- "downloaded" means bytes
|
|
107
|
+
// actually received off the wire, which is exactly the smaller,
|
|
108
|
+
// (usually) LZ4-compressed size `compress_results` exists to
|
|
109
|
+
// shrink -- see `QueryStats.bytes_downloaded`'s own doc comment
|
|
110
|
+
// in `lib.rs`.
|
|
111
|
+
stats.bytes_downloaded.fetch_add(bytes.len() as u64, Ordering::Relaxed);
|
|
112
|
+
if !compressed {
|
|
113
|
+
return Ok(bytes);
|
|
114
|
+
}
|
|
115
|
+
tokio::task::spawn_blocking(move || decompress_lz4_frame(&bytes))
|
|
116
|
+
.await
|
|
117
|
+
.map_err(join_error)?
|
|
118
|
+
})
|
|
119
|
+
.await
|
|
120
|
+
}
|
|
121
|
+
|
|
122
|
+
/// Downloads one cloud-fetch link, splitting it across parallel HTTP
|
|
123
|
+
/// Range requests when the shared `download_slots` budget has room to
|
|
124
|
+
/// spare -- see that field's own doc comment for the measured win and
|
|
125
|
+
/// why the large-result case is safe.
|
|
126
|
+
pub(crate) async fn fetch_link_bytes_budgeted(
|
|
127
|
+
self: &Arc<Self>,
|
|
128
|
+
url: &str,
|
|
129
|
+
compressed: bool,
|
|
130
|
+
stats: &Arc<QueryStatsAccumulator>,
|
|
131
|
+
) -> Result<Bytes, ApiError> {
|
|
132
|
+
// One permit per download is mandatory; the worker pool already
|
|
133
|
+
// bounds concurrent links to `chunk_fetch_concurrency`, so this
|
|
134
|
+
// never blocks in practice -- it just makes the budget accounting
|
|
135
|
+
// exact. `acquire()`'s `Err` case (the semaphore closed) can't
|
|
136
|
+
// happen -- nothing in this crate ever calls `.close()` on
|
|
137
|
+
// `download_slots` -- but propagating it as a real error instead of
|
|
138
|
+
// an `.expect()` costs nothing and avoids a panic if that ever
|
|
139
|
+
// changes.
|
|
140
|
+
let _base = self
|
|
141
|
+
.download_slots
|
|
142
|
+
.acquire()
|
|
143
|
+
.await
|
|
144
|
+
.map_err(|e| ApiError::permanent(format!("download_slots semaphore closed unexpectedly: {e}")))?;
|
|
145
|
+
let want = (MAX_SPLIT_PARTS - 1).min(self.download_slots.available_permits());
|
|
146
|
+
let extra = if want > 0 {
|
|
147
|
+
self.download_slots.try_acquire_many(want as u32).ok()
|
|
148
|
+
} else {
|
|
149
|
+
None
|
|
150
|
+
};
|
|
151
|
+
let parts = 1 + extra.as_ref().map(|p| p.num_permits()).unwrap_or(0);
|
|
152
|
+
if parts <= 1 {
|
|
153
|
+
return self.fetch_link_bytes(url, compressed, stats).await;
|
|
154
|
+
}
|
|
155
|
+
self.fetch_link_bytes_split(url, compressed, 1 << 20, parts as u64, stats)
|
|
156
|
+
.await
|
|
157
|
+
}
|
|
158
|
+
|
|
159
|
+
/// Downloads one cloud-fetch link as concurrent HTTP Range requests of
|
|
160
|
+
/// `part_size` bytes each, concatenated in order. The real object size
|
|
161
|
+
/// is learned from the first range response's `Content-Range` header --
|
|
162
|
+
/// the Thrift link's own `bytesNum` is the *uncompressed* row-set size,
|
|
163
|
+
/// not the file's size on blob storage, and using it produces HTTP 416.
|
|
164
|
+
///
|
|
165
|
+
/// A server that answers the probe with `200 OK` instead of `206
|
|
166
|
+
/// Partial Content` (Range unsupported) falls back to treating the
|
|
167
|
+
/// whole response as the complete file -- correct, since it ignored
|
|
168
|
+
/// Range and sent everything. But a server that *does* answer `206`
|
|
169
|
+
/// with a `Content-Range` header this code can't parse is a different,
|
|
170
|
+
/// unsafe case: silently treating the first `part_size` bytes as the
|
|
171
|
+
/// whole file would truncate the real result with no error. That's
|
|
172
|
+
/// treated as a hard failure instead, not a silent truncation --
|
|
173
|
+
/// confirmed unreachable against real Azure Blob Storage (always
|
|
174
|
+
/// returns a well-formed `bytes start-end/total`), but this is a
|
|
175
|
+
/// third-party response shape, not something this crate controls.
|
|
176
|
+
pub(crate) async fn fetch_link_bytes_split(
|
|
177
|
+
self: &Arc<Self>,
|
|
178
|
+
url: &str,
|
|
179
|
+
compressed: bool,
|
|
180
|
+
part_size: u64,
|
|
181
|
+
max_parts: u64,
|
|
182
|
+
stats: &Arc<QueryStatsAccumulator>,
|
|
183
|
+
) -> Result<Bytes, ApiError> {
|
|
184
|
+
// First part doubles as the size probe -- same retry_call wrapping
|
|
185
|
+
// every other download in this crate gets, so a transient failure
|
|
186
|
+
// on the probe itself doesn't skip straight to a hard error.
|
|
187
|
+
let (ranged, total, head) = self
|
|
188
|
+
.retry_call_tracked(Some(stats), || async {
|
|
189
|
+
let resp = self
|
|
190
|
+
.http
|
|
191
|
+
.get(url)
|
|
192
|
+
.header("Range", format!("bytes=0-{}", part_size - 1))
|
|
193
|
+
.timeout(self.http_timeout)
|
|
194
|
+
.send()
|
|
195
|
+
.await
|
|
196
|
+
.map_err(|e| ApiError::from_reqwest(e, true))?;
|
|
197
|
+
let status = resp.status();
|
|
198
|
+
if !status.is_success() {
|
|
199
|
+
let text = resp.text().await.unwrap_or_default();
|
|
200
|
+
return Err(ApiError::from_status(status, &text, true));
|
|
201
|
+
}
|
|
202
|
+
let ranged = status == reqwest::StatusCode::PARTIAL_CONTENT;
|
|
203
|
+
let total: Option<u64> = resp
|
|
204
|
+
.headers()
|
|
205
|
+
.get(reqwest::header::CONTENT_RANGE)
|
|
206
|
+
.and_then(|v| v.to_str().ok())
|
|
207
|
+
.and_then(|v| v.rsplit('/').next().and_then(|t| t.parse().ok()));
|
|
208
|
+
let head = resp.bytes().await.map_err(|e| ApiError::from_reqwest(e, true))?;
|
|
209
|
+
Ok((ranged, total, head))
|
|
210
|
+
})
|
|
211
|
+
.await?;
|
|
212
|
+
stats.bytes_downloaded.fetch_add(head.len() as u64, Ordering::Relaxed);
|
|
213
|
+
|
|
214
|
+
if ranged && total.is_none() {
|
|
215
|
+
return Err(ApiError::permanent(
|
|
216
|
+
"cloud-fetch link answered a Range request with 206 Partial Content but an \
|
|
217
|
+
unparseable Content-Range header -- refusing to silently return a truncated \
|
|
218
|
+
file"
|
|
219
|
+
.to_string(),
|
|
220
|
+
));
|
|
221
|
+
}
|
|
222
|
+
|
|
223
|
+
let mut handles = Vec::new();
|
|
224
|
+
if let Some(total) = ranged.then_some(total).flatten() {
|
|
225
|
+
// Spread everything after the probe part evenly over at most
|
|
226
|
+
// max_parts-1 further requests, so no single tail request
|
|
227
|
+
// dominates the wall clock.
|
|
228
|
+
let remaining = total.saturating_sub(part_size);
|
|
229
|
+
let n_rest = remaining.div_ceil(part_size).min(max_parts.saturating_sub(1));
|
|
230
|
+
let rest_size = if n_rest == 0 { 0 } else { remaining.div_ceil(n_rest) };
|
|
231
|
+
let mut start = part_size;
|
|
232
|
+
let mut n = 1u64;
|
|
233
|
+
while start < total && n <= n_rest {
|
|
234
|
+
let end = (start + rest_size - 1).min(total - 1);
|
|
235
|
+
let this = self.clone();
|
|
236
|
+
let url = url.to_string();
|
|
237
|
+
let part_stats = stats.clone();
|
|
238
|
+
handles.push(tokio::spawn(async move {
|
|
239
|
+
let bytes = this
|
|
240
|
+
.retry_call_tracked(Some(&part_stats), || async {
|
|
241
|
+
let resp = this
|
|
242
|
+
.http
|
|
243
|
+
.get(&url)
|
|
244
|
+
.header("Range", format!("bytes={start}-{end}"))
|
|
245
|
+
.timeout(this.http_timeout)
|
|
246
|
+
.send()
|
|
247
|
+
.await
|
|
248
|
+
.map_err(|e| ApiError::from_reqwest(e, true))?;
|
|
249
|
+
let status = resp.status();
|
|
250
|
+
if !status.is_success() {
|
|
251
|
+
let text = resp.text().await.unwrap_or_default();
|
|
252
|
+
return Err(ApiError::from_status(status, &text, true));
|
|
253
|
+
}
|
|
254
|
+
resp.bytes().await.map_err(|e| ApiError::from_reqwest(e, true))
|
|
255
|
+
})
|
|
256
|
+
.await;
|
|
257
|
+
if let Ok(b) = &bytes {
|
|
258
|
+
part_stats.bytes_downloaded.fetch_add(b.len() as u64, Ordering::Relaxed);
|
|
259
|
+
}
|
|
260
|
+
bytes
|
|
261
|
+
}));
|
|
262
|
+
start = end + 1;
|
|
263
|
+
n += 1;
|
|
264
|
+
}
|
|
265
|
+
}
|
|
266
|
+
|
|
267
|
+
let mut out = bytes::BytesMut::with_capacity(total.unwrap_or(head.len() as u64) as usize);
|
|
268
|
+
out.extend_from_slice(&head);
|
|
269
|
+
// Parts must concatenate in order, so this can't use `JoinSet`
|
|
270
|
+
// (which yields in completion order) without tracking indices --
|
|
271
|
+
// simpler to keep the `Vec` and explicitly `.abort()` every
|
|
272
|
+
// not-yet-awaited sibling the moment one part fails, rather than
|
|
273
|
+
// silently leaving them running (a bare `?` here would return
|
|
274
|
+
// early and just drop the rest, which does NOT cancel them --
|
|
275
|
+
// `JoinHandle::drop` detaches, it doesn't abort -- leaving up to
|
|
276
|
+
// `MAX_SPLIT_PARTS - 1` sibling Range downloads, each with their
|
|
277
|
+
// own `retry_call` backoff, still in flight for a link the caller
|
|
278
|
+
// has already given up on).
|
|
279
|
+
let mut iter = handles.into_iter();
|
|
280
|
+
while let Some(h) = iter.next() {
|
|
281
|
+
match h.await {
|
|
282
|
+
Ok(Ok(part)) => out.extend_from_slice(&part),
|
|
283
|
+
Ok(Err(e)) => {
|
|
284
|
+
for remaining in iter {
|
|
285
|
+
remaining.abort();
|
|
286
|
+
}
|
|
287
|
+
return Err(e);
|
|
288
|
+
}
|
|
289
|
+
Err(join_err) => {
|
|
290
|
+
for remaining in iter {
|
|
291
|
+
remaining.abort();
|
|
292
|
+
}
|
|
293
|
+
return Err(join_error(join_err));
|
|
294
|
+
}
|
|
295
|
+
}
|
|
296
|
+
}
|
|
297
|
+
let bytes = out.freeze();
|
|
298
|
+
if let Some(t) = total
|
|
299
|
+
&& bytes.len() as u64 != t
|
|
300
|
+
{
|
|
301
|
+
return Err(ApiError::permanent(format!(
|
|
302
|
+
"split download assembled {} bytes, expected {t}",
|
|
303
|
+
bytes.len()
|
|
304
|
+
)));
|
|
305
|
+
}
|
|
306
|
+
if !compressed {
|
|
307
|
+
return Ok(bytes);
|
|
308
|
+
}
|
|
309
|
+
tokio::task::spawn_blocking(move || decompress_lz4_frame(&bytes))
|
|
310
|
+
.await
|
|
311
|
+
.map_err(join_error)?
|
|
312
|
+
}
|
|
313
|
+
}
|
|
314
|
+
|
|
315
|
+
#[cfg(test)]
|
|
316
|
+
mod tests {
|
|
317
|
+
use super::*;
|
|
318
|
+
|
|
319
|
+
/// Regression test for a real bug found by testing against an actual
|
|
320
|
+
/// Databricks workspace (not just synthetic single-frame test data): a
|
|
321
|
+
/// real chunk's LZ4 compression is several frames concatenated back to
|
|
322
|
+
/// back (a 50k-row chunk came back as 18), and `decompress_lz4_frame`
|
|
323
|
+
/// originally only decoded the first one via one `read_to_end` call --
|
|
324
|
+
/// silently truncating to just that frame's ~200 bytes (the Arrow schema
|
|
325
|
+
/// message, no row data) with no error at all, which `pipeline/reorder.rs`'s
|
|
326
|
+
/// `decode_chunk` then happily decoded as an empty result.
|
|
327
|
+
#[test]
|
|
328
|
+
fn decompress_lz4_frame_reads_every_concatenated_frame() {
|
|
329
|
+
use std::io::Write;
|
|
330
|
+
|
|
331
|
+
fn compress_one_frame(data: &[u8]) -> Vec<u8> {
|
|
332
|
+
let mut encoder = lz4_flex::frame::FrameEncoder::new(Vec::new());
|
|
333
|
+
encoder.write_all(data).unwrap();
|
|
334
|
+
encoder.finish().unwrap()
|
|
335
|
+
}
|
|
336
|
+
|
|
337
|
+
let part_a = b"the quick brown fox jumps over the lazy dog ".repeat(50);
|
|
338
|
+
let part_b = b"pack my box with five dozen liquor jugs ".repeat(50);
|
|
339
|
+
let part_c = b"how vexingly quick daft zebras jump ".repeat(50);
|
|
340
|
+
let mut concatenated_frames = Vec::new();
|
|
341
|
+
concatenated_frames.extend(compress_one_frame(&part_a));
|
|
342
|
+
concatenated_frames.extend(compress_one_frame(&part_b));
|
|
343
|
+
concatenated_frames.extend(compress_one_frame(&part_c));
|
|
344
|
+
|
|
345
|
+
let decompressed = decompress_lz4_frame(&Bytes::from(concatenated_frames)).unwrap();
|
|
346
|
+
|
|
347
|
+
let mut expected = Vec::new();
|
|
348
|
+
expected.extend_from_slice(&part_a);
|
|
349
|
+
expected.extend_from_slice(&part_b);
|
|
350
|
+
expected.extend_from_slice(&part_c);
|
|
351
|
+
assert_eq!(decompressed, Bytes::from(expected));
|
|
352
|
+
}
|
|
353
|
+
|
|
354
|
+
/// Regression test for a bug found in code review: a real, valid LZ4
|
|
355
|
+
/// Frame that happens to decode to zero bytes (a header immediately
|
|
356
|
+
/// followed by an EndMark -- a legal frame shape, not malformed input)
|
|
357
|
+
/// makes `read_to_end` return `Ok(0)` for that frame without erroring.
|
|
358
|
+
/// The old loop read "output didn't grow" as "no more frames" and
|
|
359
|
+
/// stopped there, silently dropping every frame concatenated after the
|
|
360
|
+
/// empty one. `decompress_lz4_frame` must keep going as long as the
|
|
361
|
+
/// underlying reader still has bytes left, not just as long as output
|
|
362
|
+
/// keeps growing.
|
|
363
|
+
#[test]
|
|
364
|
+
fn decompress_lz4_frame_survives_a_zero_content_frame_in_the_middle() {
|
|
365
|
+
use std::io::Write;
|
|
366
|
+
|
|
367
|
+
fn compress_one_frame(data: &[u8]) -> Vec<u8> {
|
|
368
|
+
let mut encoder = lz4_flex::frame::FrameEncoder::new(Vec::new());
|
|
369
|
+
encoder.write_all(data).unwrap();
|
|
370
|
+
encoder.finish().unwrap()
|
|
371
|
+
}
|
|
372
|
+
|
|
373
|
+
let part_a = b"the quick brown fox jumps over the lazy dog ".repeat(50);
|
|
374
|
+
let part_b = b"pack my box with five dozen liquor jugs ".repeat(50);
|
|
375
|
+
let mut concatenated_frames = Vec::new();
|
|
376
|
+
concatenated_frames.extend(compress_one_frame(&part_a));
|
|
377
|
+
concatenated_frames.extend(compress_one_frame(b"")); // real frame, zero content
|
|
378
|
+
concatenated_frames.extend(compress_one_frame(&part_b));
|
|
379
|
+
|
|
380
|
+
let decompressed = decompress_lz4_frame(&Bytes::from(concatenated_frames)).unwrap();
|
|
381
|
+
|
|
382
|
+
let mut expected = Vec::new();
|
|
383
|
+
expected.extend_from_slice(&part_a);
|
|
384
|
+
expected.extend_from_slice(&part_b);
|
|
385
|
+
assert_eq!(
|
|
386
|
+
decompressed,
|
|
387
|
+
Bytes::from(expected),
|
|
388
|
+
"the frame after the zero-content one must not be silently dropped"
|
|
389
|
+
);
|
|
390
|
+
}
|
|
391
|
+
|
|
392
|
+
/// Regression test for a real bug found against a real workspace: a
|
|
393
|
+
/// 400-chunk/5.6M-row fetch failed permanently on one of many large
|
|
394
|
+
/// concurrent blob downloads with reqwest's "error decoding response
|
|
395
|
+
/// body" -- a connection that closes early mid-body (`Kind::Decode`,
|
|
396
|
+
/// see `ApiError::from_reqwest`'s comment), not a genuinely dead
|
|
397
|
+
/// endpoint. Before the fix, `transient` was unconditionally `false` for
|
|
398
|
+
/// every reqwest error, so `retry_call` never got a second attempt and
|
|
399
|
+
/// the whole query failed outright. A raw truncated-response server
|
|
400
|
+
/// (rather than wiremock, which doesn't expose a way to violate its own
|
|
401
|
+
/// Content-Length) reproduces the same client-side error reqwest raised
|
|
402
|
+
/// against the real blob storage endpoint.
|
|
403
|
+
#[tokio::test]
|
|
404
|
+
async fn fetch_link_bytes_retries_after_a_connection_closed_mid_body() {
|
|
405
|
+
use tokio::io::{AsyncReadExt, AsyncWriteExt};
|
|
406
|
+
use tokio::net::TcpListener;
|
|
407
|
+
|
|
408
|
+
let listener = TcpListener::bind("127.0.0.1:0").await.unwrap();
|
|
409
|
+
let addr = listener.local_addr().unwrap();
|
|
410
|
+
|
|
411
|
+
tokio::spawn(async move {
|
|
412
|
+
let mut attempt = 0u32;
|
|
413
|
+
loop {
|
|
414
|
+
let Ok((mut socket, _)) = listener.accept().await else {
|
|
415
|
+
return;
|
|
416
|
+
};
|
|
417
|
+
attempt += 1;
|
|
418
|
+
let mut buf = [0u8; 1024];
|
|
419
|
+
let _ = socket.read(&mut buf).await;
|
|
420
|
+
if attempt == 1 {
|
|
421
|
+
// Claims 100 bytes, sends 10, then closes -- reqwest's
|
|
422
|
+
// `.bytes()` surfaces exactly this as `Kind::Decode`.
|
|
423
|
+
let _ = socket
|
|
424
|
+
.write_all(b"HTTP/1.1 200 OK\r\nContent-Length: 100\r\n\r\n0123456789")
|
|
425
|
+
.await;
|
|
426
|
+
let _ = socket.shutdown().await;
|
|
427
|
+
} else {
|
|
428
|
+
let _ = socket
|
|
429
|
+
.write_all(b"HTTP/1.1 200 OK\r\nContent-Length: 5\r\n\r\nhello")
|
|
430
|
+
.await;
|
|
431
|
+
let _ = socket.shutdown().await;
|
|
432
|
+
return;
|
|
433
|
+
}
|
|
434
|
+
}
|
|
435
|
+
});
|
|
436
|
+
|
|
437
|
+
let client = DbClient::new(&format!("http://{addr}"), "wh-test", "fake-token");
|
|
438
|
+
let stats = QueryStatsAccumulator::default();
|
|
439
|
+
let bytes = client
|
|
440
|
+
.fetch_link_bytes(&format!("http://{addr}/data"), false, &stats)
|
|
441
|
+
.await
|
|
442
|
+
.expect("must retry past the truncated first attempt and succeed on the second");
|
|
443
|
+
assert_eq!(&bytes[..], b"hello");
|
|
444
|
+
assert_eq!(
|
|
445
|
+
stats.retry_count.load(Ordering::Relaxed),
|
|
446
|
+
1,
|
|
447
|
+
"the one retry after the truncated first attempt must be counted"
|
|
448
|
+
);
|
|
449
|
+
assert_eq!(bytes.len() as u64, stats.bytes_downloaded.load(Ordering::Relaxed));
|
|
450
|
+
}
|
|
451
|
+
}
|
|
452
|
+
|
|
453
|
+
/// Property-based fuzzing for `decompress_lz4_frame` -- the other real,
|
|
454
|
+
/// previously-shipped bug class this crate's hand-rolled/third-party-library-
|
|
455
|
+
/// adjacent parsing code has already produced (the multi-frame truncation
|
|
456
|
+
/// bug documented on this function's own call site in `execute_statement`'s
|
|
457
|
+
/// doc comment, and `decompress_lz4_frame_survives_a_zero_content_frame_in_the_middle`
|
|
458
|
+
/// above -- both found by real-workspace testing, not fuzzing). Separate
|
|
459
|
+
/// `#[cfg(test)]` module from `mod tests` above for the same reason
|
|
460
|
+
/// `thrift.rs`/`json_convert.rs`'s own `mod proptests` are split out.
|
|
461
|
+
#[cfg(test)]
|
|
462
|
+
mod proptests {
|
|
463
|
+
use proptest::prelude::*;
|
|
464
|
+
|
|
465
|
+
use super::*;
|
|
466
|
+
|
|
467
|
+
proptest! {
|
|
468
|
+
/// Arbitrary/malformed/truncated/empty bytes reinterpreted as an LZ4
|
|
469
|
+
/// Frame: must never panic (most inputs are simply not a valid LZ4
|
|
470
|
+
/// Frame at all and should return `Err`; a few short/empty inputs
|
|
471
|
+
/// are legal-but-trivial frames and should return `Ok` with little
|
|
472
|
+
/// or no content -- either outcome is fine, only a panic is a bug).
|
|
473
|
+
#[test]
|
|
474
|
+
fn decompress_lz4_frame_never_panics_on_arbitrary_bytes(bytes in proptest::collection::vec(any::<u8>(), 0..4096)) {
|
|
475
|
+
let _ = decompress_lz4_frame(&Bytes::from(bytes));
|
|
476
|
+
}
|
|
477
|
+
|
|
478
|
+
/// A *truncated* real frame -- compress real data, then chop the
|
|
479
|
+
/// tail off at an arbitrary point -- targeting the multi-frame loop
|
|
480
|
+
/// specifically (a cut mid-frame, mid-block, or exactly on a frame
|
|
481
|
+
/// boundary), which arbitrary random bytes essentially never
|
|
482
|
+
/// produce (LZ4 Frame's magic number alone is 4 specific bytes).
|
|
483
|
+
#[test]
|
|
484
|
+
fn decompress_lz4_frame_never_panics_on_a_truncated_real_frame(
|
|
485
|
+
payload in proptest::collection::vec(any::<u8>(), 0..2048),
|
|
486
|
+
cut_at_fraction in 0.0f64..=1.0,
|
|
487
|
+
) {
|
|
488
|
+
use std::io::Write;
|
|
489
|
+
let mut encoder = lz4_flex::frame::FrameEncoder::new(Vec::new());
|
|
490
|
+
encoder.write_all(&payload).unwrap();
|
|
491
|
+
let full = encoder.finish().unwrap();
|
|
492
|
+
let cut = ((full.len() as f64) * cut_at_fraction) as usize;
|
|
493
|
+
let _ = decompress_lz4_frame(&Bytes::from(full[..cut].to_vec()));
|
|
494
|
+
}
|
|
495
|
+
}
|
|
496
|
+
}
|