arrowbricks 3.1.2__tar.gz → 3.1.3__tar.gz

This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
Files changed (48) hide show
  1. {arrowbricks-3.1.2 → arrowbricks-3.1.3}/PKG-INFO +30 -6
  2. {arrowbricks-3.1.2 → arrowbricks-3.1.3}/README.md +30 -6
  3. {arrowbricks-3.1.2 → arrowbricks-3.1.3}/pyproject.toml +1 -1
  4. {arrowbricks-3.1.2 → arrowbricks-3.1.3}/rust/arrowbricks_core/Cargo.lock +4 -64
  5. {arrowbricks-3.1.2 → arrowbricks-3.1.3}/rust/arrowbricks_core/Cargo.toml +18 -5
  6. {arrowbricks-3.1.2 → arrowbricks-3.1.3}/rust/arrowbricks_core/README.md +6 -0
  7. {arrowbricks-3.1.2 → arrowbricks-3.1.3}/rust/arrowbricks_core/src/client/download.rs +151 -101
  8. {arrowbricks-3.1.2 → arrowbricks-3.1.3}/rust/arrowbricks_core/src/client.rs +2 -2
  9. {arrowbricks-3.1.2 → arrowbricks-3.1.3}/rust/arrowbricks_core/src/json_convert.rs +8 -7
  10. {arrowbricks-3.1.2 → arrowbricks-3.1.3}/rust/arrowbricks_core/src/lib.rs +9 -10
  11. {arrowbricks-3.1.2 → arrowbricks-3.1.3}/rust/arrowbricks_core/src/pipeline/ndjson.rs +9 -7
  12. {arrowbricks-3.1.2 → arrowbricks-3.1.3}/rust/arrowbricks_core/src/pipeline/reorder.rs +33 -20
  13. {arrowbricks-3.1.2 → arrowbricks-3.1.3}/rust/arrowbricks_core/src/pipeline/sea.rs +2 -2
  14. {arrowbricks-3.1.2 → arrowbricks-3.1.3}/rust/arrowbricks_core/src/pipeline/test_support.rs +3 -3
  15. {arrowbricks-3.1.2 → arrowbricks-3.1.3}/rust/arrowbricks_core/src/pipeline/thrift_exec.rs +27 -20
  16. {arrowbricks-3.1.2 → arrowbricks-3.1.3}/rust/arrowbricks_core/src/pipeline.rs +2 -0
  17. {arrowbricks-3.1.2 → arrowbricks-3.1.3}/rust/arrowbricks_core/src/thrift.rs +3 -3
  18. {arrowbricks-3.1.2 → arrowbricks-3.1.3}/rust/arrowbricks_core/tests/wiremock_pipeline.rs +4 -4
  19. {arrowbricks-3.1.2 → arrowbricks-3.1.3}/rust/arrowbricks_core/tests/wiremock_thrift.rs +118 -5
  20. {arrowbricks-3.1.2 → arrowbricks-3.1.3}/rust/arrowbricks_core/tests_py/test_ipc_stream.py +57 -0
  21. {arrowbricks-3.1.2 → arrowbricks-3.1.3}/LICENSE +0 -0
  22. {arrowbricks-3.1.2 → arrowbricks-3.1.3}/rust/arrowbricks_core/.gitignore +0 -0
  23. {arrowbricks-3.1.2 → arrowbricks-3.1.3}/rust/arrowbricks_core/examples/duckdb_query.py +0 -0
  24. {arrowbricks-3.1.2 → arrowbricks-3.1.3}/rust/arrowbricks_core/examples/fastapi_sse.py +0 -0
  25. {arrowbricks-3.1.2 → arrowbricks-3.1.3}/rust/arrowbricks_core/rustfmt.toml +0 -0
  26. {arrowbricks-3.1.2 → arrowbricks-3.1.3}/rust/arrowbricks_core/src/client/error.rs +0 -0
  27. {arrowbricks-3.1.2 → arrowbricks-3.1.3}/rust/arrowbricks_core/src/client/model.rs +0 -0
  28. {arrowbricks-3.1.2 → arrowbricks-3.1.3}/rust/arrowbricks_core/src/client/sea.rs +0 -0
  29. {arrowbricks-3.1.2 → arrowbricks-3.1.3}/rust/arrowbricks_core/src/client/thrift_rpc.rs +0 -0
  30. {arrowbricks-3.1.2 → arrowbricks-3.1.3}/rust/arrowbricks_core/src/client/volume.rs +0 -0
  31. {arrowbricks-3.1.2 → arrowbricks-3.1.3}/rust/arrowbricks_core/src/heartbeat.rs +0 -0
  32. {arrowbricks-3.1.2 → arrowbricks-3.1.3}/rust/arrowbricks_core/src/pipeline/stats.rs +0 -0
  33. {arrowbricks-3.1.2 → arrowbricks-3.1.3}/rust/arrowbricks_core/tests/common/mod.rs +0 -0
  34. {arrowbricks-3.1.2 → arrowbricks-3.1.3}/rust/arrowbricks_core/tests/wiremock_volume_files.rs +0 -0
  35. {arrowbricks-3.1.2 → arrowbricks-3.1.3}/rust/arrowbricks_core/tests_py/conftest.py +0 -0
  36. {arrowbricks-3.1.2 → arrowbricks-3.1.3}/rust/arrowbricks_core/tests_py/test_parameters.py +0 -0
  37. {arrowbricks-3.1.2 → arrowbricks-3.1.3}/rust/arrowbricks_core/tests_py/test_stream_ndjson_lines.py +0 -0
  38. {arrowbricks-3.1.2 → arrowbricks-3.1.3}/rust/arrowbricks_core/tests_py/test_streaming.py +0 -0
  39. {arrowbricks-3.1.2 → arrowbricks-3.1.3}/rust/arrowbricks_core/tests_py/test_thrift.py +0 -0
  40. {arrowbricks-3.1.2 → arrowbricks-3.1.3}/rust/arrowbricks_core/tests_py/test_token_provider.py +0 -0
  41. {arrowbricks-3.1.2 → arrowbricks-3.1.3}/rust/arrowbricks_core/tests_py/test_volume_files.py +0 -0
  42. {arrowbricks-3.1.2 → arrowbricks-3.1.3}/rust/arrowbricks_core/tests_py/thrift_mock.py +0 -0
  43. {arrowbricks-3.1.2 → arrowbricks-3.1.3}/src/arrowbricks/__init__.py +0 -0
  44. {arrowbricks-3.1.2 → arrowbricks-3.1.3}/src/arrowbricks/_core.pyi +0 -0
  45. {arrowbricks-3.1.2 → arrowbricks-3.1.3}/src/arrowbricks/_streaming.py +0 -0
  46. {arrowbricks-3.1.2 → arrowbricks-3.1.3}/src/arrowbricks/client.py +0 -0
  47. {arrowbricks-3.1.2 → arrowbricks-3.1.3}/src/arrowbricks/cursor.py +0 -0
  48. {arrowbricks-3.1.2 → arrowbricks-3.1.3}/src/arrowbricks/py.typed +0 -0
@@ -1,6 +1,6 @@
1
1
  Metadata-Version: 2.4
2
2
  Name: arrowbricks
3
- Version: 3.1.2
3
+ Version: 3.1.3
4
4
  Requires-Dist: arro3-core>=0.8 ; extra == 'arro3'
5
5
  Provides-Extra: arro3
6
6
  License-File: LICENSE
@@ -271,12 +271,17 @@ The [official driver](https://github.com/databricks/databricks-sql-python) is th
271
271
 
272
272
  Measured against a real Databricks SQL warehouse (Azure Databricks, `2X-Small` **Pro** serverless warehouse, Photon on, 1-4 auto-scaling clusters -- the smallest/cheapest warehouse tier, deliberately: a bigger warehouse would narrow the gap by making the query itself slower and the client-side overhead this compares proportionally smaller). Query: `SELECT id, id * 2 AS doubled, CAST(id AS STRING) AS label FROM range(200000)` (200k rows, 3 columns), 3 timed runs after 1 discarded warm-up run, one connection reused per library:
273
273
 
274
- | | avg | stdev | range | peak RSS during the query |
275
- |---|---|---|---|---|
276
- | `databricks-sql-connector` | 0.88s | 0.06s | 0.83s - 0.95s | 16 MB |
277
- | arrowbricks | 0.50s | 0.02s | 0.48s - 0.52s | 16 MB |
274
+ | | avg | stdev | range |
275
+ |---|---|---|---|
276
+ | `databricks-sql-connector` | 0.88s | 0.06s | 0.83s - 0.95s |
277
+ | arrowbricks | 0.50s | 0.02s | 0.48s - 0.52s |
278
278
 
279
- **arrowbricks: ~1.8x faster**, same order-of-magnitude peak memory *for this query size* -- at 200k rows the Python interpreter's own baseline footprint dominates over the actual result data for both libraries, so this particular number doesn't show a difference. The real memory/footprint difference is in what gets installed, not what a single small query allocates:
279
+ These historical latency measurements compare the connector's `fetchall()`
280
+ (Python rows) with arrowbricks' `fetchall_arrow()` (Arrow), which do different
281
+ amounts of materialization. The previously published 16 MB memory figures
282
+ were invalid: the script sampled peak RSS before importing the libraries or
283
+ running queries. It now measures after the workload; rerun it for actual
284
+ peak memory on your machine. The installed-footprint measurements were:
280
285
 
281
286
  | | installed size (package + all required deps) |
282
287
  |---|---|
@@ -292,6 +297,25 @@ DATABRICKS_TOKEN=dapiXXXXXXXXXXXXXXXXXXXXXXXXXXXX \
292
297
  python examples/benchmark_vs_connector.py
293
298
  ```
294
299
 
300
+ For comparing arrowbricks versions with identical APIs, use
301
+ [`examples/benchmark_versions.py`](examples/benchmark_versions.py). It keeps
302
+ one connection per version, discards warm-ups, alternates execution order,
303
+ passes concurrency explicitly, and checks row counts. See
304
+ [`benchmarks/2026-09-06.md`](benchmarks/2026-09-06.md) for replay/cache measurements
305
+ and [`benchmarks/2026-09-06-downloads.md`](benchmarks/2026-09-06-downloads.md)
306
+ for the subsequent cloud-fetch scheduling measurements and limits.
307
+
308
+ Cached IPC replay can be measured without a warehouse:
309
+
310
+ ```bash
311
+ uv run python examples/benchmark_replay.py --rows 200000 --columns 16 --repeats 8
312
+ ```
313
+
314
+ Replays now share immutable input bytes across decoded tables, reducing
315
+ repeated copies and memory use. Arrow may still copy misaligned fixed-width
316
+ buffers or decompress IPC-compressed bodies. Keeping a small slice of a
317
+ decoded array can retain the full source allocation until that slice is released.
318
+
295
319
  `BENCHMARK_SQL` overrides the query, `BENCHMARK_RUNS` (default 3) controls how many timed runs to average.
296
320
 
297
321
  ## A note on Arrow IPC compression
@@ -259,12 +259,17 @@ The [official driver](https://github.com/databricks/databricks-sql-python) is th
259
259
 
260
260
  Measured against a real Databricks SQL warehouse (Azure Databricks, `2X-Small` **Pro** serverless warehouse, Photon on, 1-4 auto-scaling clusters -- the smallest/cheapest warehouse tier, deliberately: a bigger warehouse would narrow the gap by making the query itself slower and the client-side overhead this compares proportionally smaller). Query: `SELECT id, id * 2 AS doubled, CAST(id AS STRING) AS label FROM range(200000)` (200k rows, 3 columns), 3 timed runs after 1 discarded warm-up run, one connection reused per library:
261
261
 
262
- | | avg | stdev | range | peak RSS during the query |
263
- |---|---|---|---|---|
264
- | `databricks-sql-connector` | 0.88s | 0.06s | 0.83s - 0.95s | 16 MB |
265
- | arrowbricks | 0.50s | 0.02s | 0.48s - 0.52s | 16 MB |
266
-
267
- **arrowbricks: ~1.8x faster**, same order-of-magnitude peak memory *for this query size* -- at 200k rows the Python interpreter's own baseline footprint dominates over the actual result data for both libraries, so this particular number doesn't show a difference. The real memory/footprint difference is in what gets installed, not what a single small query allocates:
262
+ | | avg | stdev | range |
263
+ |---|---|---|---|
264
+ | `databricks-sql-connector` | 0.88s | 0.06s | 0.83s - 0.95s |
265
+ | arrowbricks | 0.50s | 0.02s | 0.48s - 0.52s |
266
+
267
+ These historical latency measurements compare the connector's `fetchall()`
268
+ (Python rows) with arrowbricks' `fetchall_arrow()` (Arrow), which do different
269
+ amounts of materialization. The previously published 16 MB memory figures
270
+ were invalid: the script sampled peak RSS before importing the libraries or
271
+ running queries. It now measures after the workload; rerun it for actual
272
+ peak memory on your machine. The installed-footprint measurements were:
268
273
 
269
274
  | | installed size (package + all required deps) |
270
275
  |---|---|
@@ -280,6 +285,25 @@ DATABRICKS_TOKEN=dapiXXXXXXXXXXXXXXXXXXXXXXXXXXXX \
280
285
  python examples/benchmark_vs_connector.py
281
286
  ```
282
287
 
288
+ For comparing arrowbricks versions with identical APIs, use
289
+ [`examples/benchmark_versions.py`](examples/benchmark_versions.py). It keeps
290
+ one connection per version, discards warm-ups, alternates execution order,
291
+ passes concurrency explicitly, and checks row counts. See
292
+ [`benchmarks/2026-09-06.md`](benchmarks/2026-09-06.md) for replay/cache measurements
293
+ and [`benchmarks/2026-09-06-downloads.md`](benchmarks/2026-09-06-downloads.md)
294
+ for the subsequent cloud-fetch scheduling measurements and limits.
295
+
296
+ Cached IPC replay can be measured without a warehouse:
297
+
298
+ ```bash
299
+ uv run python examples/benchmark_replay.py --rows 200000 --columns 16 --repeats 8
300
+ ```
301
+
302
+ Replays now share immutable input bytes across decoded tables, reducing
303
+ repeated copies and memory use. Arrow may still copy misaligned fixed-width
304
+ buffers or decompress IPC-compressed bodies. Keeping a small slice of a
305
+ decoded array can retain the full source allocation until that slice is released.
306
+
283
307
  `BENCHMARK_SQL` overrides the query, `BENCHMARK_RUNS` (default 3) controls how many timed runs to average.
284
308
 
285
309
  ## A note on Arrow IPC compression
@@ -1,6 +1,6 @@
1
1
  [project]
2
2
  name = "arrowbricks"
3
- version = "3.1.2"
3
+ version = "3.1.3"
4
4
  description = "Runs SQL against a Databricks SQL warehouse via the Statement Execution API and hands you the result as Arrow -- a DB-API-ish Cursor (fetchone/fetchmany/fetchall/fetchall_arrow) or NDJSON streaming. Rust/PyO3 core throughout -- zero required runtime dependencies."
5
5
  readme = "README.md"
6
6
  license = "MIT"
@@ -34,39 +34,6 @@ dependencies = [
34
34
  "libc",
35
35
  ]
36
36
 
37
- [[package]]
38
- name = "arrow"
39
- version = "59.2.0"
40
- source = "registry+https://github.com/rust-lang/crates.io-index"
41
- checksum = "61d285d16bce7d0be61912f7928342b673067b6b7d7ef6cc179258ba7de1fecf"
42
- dependencies = [
43
- "arrow-arith",
44
- "arrow-array",
45
- "arrow-buffer",
46
- "arrow-cast",
47
- "arrow-data",
48
- "arrow-ipc",
49
- "arrow-ord",
50
- "arrow-row",
51
- "arrow-schema",
52
- "arrow-select",
53
- "arrow-string",
54
- ]
55
-
56
- [[package]]
57
- name = "arrow-arith"
58
- version = "59.2.0"
59
- source = "registry+https://github.com/rust-lang/crates.io-index"
60
- checksum = "757ef1836251e88222542a7da2623bc1c9cb9e20afefa6db2c41e79991cd91d4"
61
- dependencies = [
62
- "arrow-array",
63
- "arrow-buffer",
64
- "arrow-data",
65
- "arrow-schema",
66
- "chrono",
67
- "num-traits",
68
- ]
69
-
70
37
  [[package]]
71
38
  name = "arrow-array"
72
39
  version = "59.2.0"
@@ -186,19 +153,6 @@ dependencies = [
186
153
  "arrow-select",
187
154
  ]
188
155
 
189
- [[package]]
190
- name = "arrow-row"
191
- version = "59.2.0"
192
- source = "registry+https://github.com/rust-lang/crates.io-index"
193
- checksum = "bbec439386df71ad570e6758a946111322b9e9dc8db83b5527321f0b4c9119c2"
194
- dependencies = [
195
- "arrow-array",
196
- "arrow-buffer",
197
- "arrow-data",
198
- "arrow-schema",
199
- "half",
200
- ]
201
-
202
156
  [[package]]
203
157
  name = "arrow-schema"
204
158
  version = "59.2.0"
@@ -223,28 +177,14 @@ dependencies = [
223
177
  ]
224
178
 
225
179
  [[package]]
226
- name = "arrow-string"
227
- version = "59.2.0"
228
- source = "registry+https://github.com/rust-lang/crates.io-index"
229
- checksum = "c838a25bb3691e919e0f617616ac51a4ff8517a952e29ca133cf0c22b2ce65b1"
180
+ name = "arrowbricks_core"
181
+ version = "3.1.3"
230
182
  dependencies = [
231
183
  "arrow-array",
232
184
  "arrow-buffer",
233
- "arrow-data",
234
- "arrow-schema",
235
- "arrow-select",
236
- "memchr",
237
- "num-traits",
238
- "regex",
239
- "regex-syntax",
240
- ]
241
-
242
- [[package]]
243
- name = "arrowbricks_core"
244
- version = "3.1.2"
245
- dependencies = [
246
- "arrow",
185
+ "arrow-ipc",
247
186
  "arrow-json",
187
+ "arrow-schema",
248
188
  "base64 0.23.1",
249
189
  "bytes",
250
190
  "chrono",
@@ -1,6 +1,6 @@
1
1
  [package]
2
2
  name = "arrowbricks_core"
3
- version = "3.1.2"
3
+ version = "3.1.3"
4
4
  edition = "2024"
5
5
  readme = "README.md"
6
6
 
@@ -9,10 +9,12 @@ name = "arrowbricks_core"
9
9
  crate-type = ["cdylib", "rlib"]
10
10
 
11
11
  [dependencies]
12
- # default-features off: arrow's defaults are csv+ipc+json, and we only ever
13
- # decode Arrow-IPC chunks -- csv/json parsing (plus their transitive deps)
14
- # would just be dead weight in the shipped .so.
15
- arrow = { version = "59.1.0", default-features = false, features = ["ipc"] }
12
+ # Depend on the components we use. The umbrella `arrow` crate also builds
13
+ # arithmetic, string, and row kernels even with default-features disabled.
14
+ arrow-array = "59.1.0"
15
+ arrow-buffer = "59.1.0"
16
+ arrow-ipc = "59.1.0"
17
+ arrow-schema = "59.1.0"
16
18
  arrow-json = "59.1.0"
17
19
  bytes = "1.12.1"
18
20
  # Already an unconditional transitive dependency of arrow-cast (pulled in by
@@ -103,6 +105,11 @@ redundant_clone = "warn"
103
105
  large_enum_variant = "warn"
104
106
  needless_collect = "warn"
105
107
 
108
+ [profile.dev]
109
+ # Keep source locations in backtraces without caching full debug type data
110
+ # for Arrow and the HTTP stack. Incremental compilation remains enabled.
111
+ debug = "line-tables-only"
112
+
106
113
  [profile.release]
107
114
  # opt-level="s" over the implicit default (3) or "z" -- a size-vs-speed
108
115
  # tradeoff previously rejected here on the reasoning that "3" was what the
@@ -126,3 +133,9 @@ opt-level = "s"
126
133
  strip = true
127
134
  lto = true
128
135
  codegen-units = 1
136
+
137
+ [profile.release.package.pyo3-arrow]
138
+ # This bridge is dominated by interoperability helpers. Optimizing it for
139
+ # size saves ~32 KiB on macOS arm64; the IPC decoder keeps opt-level="s".
140
+ # Replay and live-query benchmarks are recorded in benchmarks/2026-09-06.md.
141
+ opt-level = "z"
@@ -32,6 +32,12 @@ This builds a mixed Python/Rust project: the compiled extension lands at
32
32
  `mypy`/`pyright`/`ty` type-checking -- a compiled PyO3 extension has no type
33
33
  info of its own without them.
34
34
 
35
+ Development/test builds retain source line information for backtraces but
36
+ omit full debug type data to reduce the Cargo cache. For full debugger
37
+ variable inspection, run with `CARGO_PROFILE_DEV_DEBUG=2`. Incremental
38
+ compilation stays enabled. Release builds optimize the Python/Arrow bridge
39
+ for size while keeping the IPC decoder at its existing optimization level.
40
+
35
41
  ## Quickstart
36
42
 
37
43
  Importable directly (bypassing the `arrowbricks` Python wrapper) as
@@ -121,28 +121,24 @@ impl DbClient {
121
121
 
122
122
  /// Downloads one cloud-fetch link, splitting it across parallel HTTP
123
123
  /// Range requests when the shared `download_slots` budget has room to
124
- /// spare -- see that field's own doc comment for the measured win and
125
- /// why the large-result case is safe.
124
+ /// spare. `split_limit` reserves a fair share for the other links in
125
+ /// the same discovered batch; available permits alone do not tell us
126
+ /// how many workers have yet to start requesting their own slots.
126
127
  pub(crate) async fn fetch_link_bytes_budgeted(
127
128
  self: &Arc<Self>,
128
129
  url: &str,
129
130
  compressed: bool,
131
+ split_limit: usize,
130
132
  stats: &Arc<QueryStatsAccumulator>,
131
133
  ) -> Result<Bytes, ApiError> {
132
- // One permit per download is mandatory; the worker pool already
133
- // bounds concurrent links to `chunk_fetch_concurrency`, so this
134
- // never blocks in practice -- it just makes the budget accounting
135
- // exact. `acquire()`'s `Err` case (the semaphore closed) can't
136
- // happen -- nothing in this crate ever calls `.close()` on
137
- // `download_slots` -- but propagating it as a real error instead of
138
- // an `.expect()` costs nothing and avoids a panic if that ever
139
- // changes.
134
+ // One permit per file is mandatory. This can wait because other
135
+ // files' extra Range requests consume the same shared budget.
140
136
  let _base = self
141
137
  .download_slots
142
138
  .acquire()
143
139
  .await
144
140
  .map_err(|e| ApiError::permanent(format!("download_slots semaphore closed unexpectedly: {e}")))?;
145
- let want = (MAX_SPLIT_PARTS - 1).min(self.download_slots.available_permits());
141
+ let want = (MAX_SPLIT_PARTS.min(split_limit.max(1)) - 1).min(self.download_slots.available_permits());
146
142
  let extra = if want > 0 {
147
143
  self.download_slots.try_acquire_many(want as u32).ok()
148
144
  } else {
@@ -156,8 +152,8 @@ impl DbClient {
156
152
  .await
157
153
  }
158
154
 
159
- /// Downloads one cloud-fetch link as concurrent HTTP Range requests of
160
- /// `part_size` bytes each, concatenated in order. The real object size
155
+ /// Downloads a `part_size` probe and spreads the remainder across
156
+ /// bounded parallel Range requests, concatenated in order. The real object size
161
157
  /// is learned from the first range response's `Content-Range` header --
162
158
  /// the Thrift link's own `bytesNum` is the *uncompressed* row-set size,
163
159
  /// not the file's size on blob storage, and using it produces HTTP 416.
@@ -181,10 +177,11 @@ impl DbClient {
181
177
  max_parts: u64,
182
178
  stats: &Arc<QueryStatsAccumulator>,
183
179
  ) -> Result<Bytes, ApiError> {
184
- // First part doubles as the size probe -- same retry_call wrapping
185
- // every other download in this crate gets, so a transient failure
186
- // on the probe itself doesn't skip straight to a hard error.
187
- let (ranged, total, head) = self
180
+ // Retry the probe as one request (headers and body). Start tail
181
+ // downloads as soon as its headers reveal the file size, so reading
182
+ // the first MiB overlaps the rest. JoinSet aborts those requests if
183
+ // the probe fails and retries, or the surrounding fetch is cancelled.
184
+ let (total, head, mut downloads) = self
188
185
  .retry_call_tracked(Some(stats), || async {
189
186
  let resp = self
190
187
  .http
@@ -205,96 +202,84 @@ impl DbClient {
205
202
  .get(reqwest::header::CONTENT_RANGE)
206
203
  .and_then(|v| v.to_str().ok())
207
204
  .and_then(|v| v.rsplit('/').next().and_then(|t| t.parse().ok()));
208
- let head = resp.bytes().await.map_err(|e| ApiError::from_reqwest(e, true))?;
209
- Ok((ranged, total, head))
210
- })
211
- .await?;
212
- stats.bytes_downloaded.fetch_add(head.len() as u64, Ordering::Relaxed);
213
-
214
- if ranged && total.is_none() {
215
- return Err(ApiError::permanent(
216
- "cloud-fetch link answered a Range request with 206 Partial Content but an \
217
- unparseable Content-Range header -- refusing to silently return a truncated \
218
- file"
219
- .to_string(),
220
- ));
221
- }
205
+ if ranged && total.is_none() {
206
+ return Err(ApiError::permanent(
207
+ "cloud-fetch link answered a Range request with 206 Partial Content but an \
208
+ unparseable Content-Range header -- refusing to silently return a truncated file",
209
+ ));
210
+ }
222
211
 
223
- let mut handles = Vec::new();
224
- if let Some(total) = ranged.then_some(total).flatten() {
225
- // Spread everything after the probe part evenly over at most
226
- // max_parts-1 further requests, so no single tail request
227
- // dominates the wall clock.
228
- let remaining = total.saturating_sub(part_size);
229
- let n_rest = remaining.div_ceil(part_size).min(max_parts.saturating_sub(1));
230
- let rest_size = if n_rest == 0 { 0 } else { remaining.div_ceil(n_rest) };
231
- let mut start = part_size;
232
- let mut n = 1u64;
233
- while start < total && n <= n_rest {
234
- let end = (start + rest_size - 1).min(total - 1);
235
- let this = self.clone();
236
- let url = url.to_string();
237
- let part_stats = stats.clone();
238
- handles.push(tokio::spawn(async move {
239
- let bytes = this
240
- .retry_call_tracked(Some(&part_stats), || async {
241
- let resp = this
242
- .http
243
- .get(&url)
244
- .header("Range", format!("bytes={start}-{end}"))
245
- .timeout(this.http_timeout)
246
- .send()
247
- .await
248
- .map_err(|e| ApiError::from_reqwest(e, true))?;
249
- let status = resp.status();
250
- if !status.is_success() {
251
- let text = resp.text().await.unwrap_or_default();
252
- return Err(ApiError::from_status(status, &text, true));
212
+ let mut downloads = tokio::task::JoinSet::new();
213
+ if let Some(total) = ranged.then_some(total).flatten() {
214
+ let remaining = total.saturating_sub(part_size);
215
+ let n_rest = remaining.div_ceil(part_size).min(max_parts.saturating_sub(1));
216
+ let rest_size = if n_rest == 0 { 0 } else { remaining.div_ceil(n_rest) };
217
+ let mut start = part_size;
218
+ for index in 0..n_rest {
219
+ if start >= total {
220
+ break;
221
+ }
222
+ let end = (start + rest_size - 1).min(total - 1);
223
+ let this = self.clone();
224
+ let url = url.to_string();
225
+ let part_stats = stats.clone();
226
+ downloads.spawn(async move {
227
+ let bytes = this
228
+ .retry_call_tracked(Some(&part_stats), || async {
229
+ let resp = this
230
+ .http
231
+ .get(&url)
232
+ .header("Range", format!("bytes={start}-{end}"))
233
+ .timeout(this.http_timeout)
234
+ .send()
235
+ .await
236
+ .map_err(|e| ApiError::from_reqwest(e, true))?;
237
+ let status = resp.status();
238
+ if !status.is_success() {
239
+ let text = resp.text().await.unwrap_or_default();
240
+ return Err(ApiError::from_status(status, &text, true));
241
+ }
242
+ resp.bytes().await.map_err(|e| ApiError::from_reqwest(e, true))
243
+ })
244
+ .await;
245
+ if let Ok(b) = &bytes {
246
+ part_stats.bytes_downloaded.fetch_add(b.len() as u64, Ordering::Relaxed);
253
247
  }
254
- resp.bytes().await.map_err(|e| ApiError::from_reqwest(e, true))
255
- })
256
- .await;
257
- if let Ok(b) = &bytes {
258
- part_stats.bytes_downloaded.fetch_add(b.len() as u64, Ordering::Relaxed);
259
- }
260
- bytes
261
- }));
262
- start = end + 1;
263
- n += 1;
264
- }
265
- }
266
-
267
- let mut out = bytes::BytesMut::with_capacity(total.unwrap_or(head.len() as u64) as usize);
268
- out.extend_from_slice(&head);
269
- // Parts must concatenate in order, so this can't use `JoinSet`
270
- // (which yields in completion order) without tracking indices --
271
- // simpler to keep the `Vec` and explicitly `.abort()` every
272
- // not-yet-awaited sibling the moment one part fails, rather than
273
- // silently leaving them running (a bare `?` here would return
274
- // early and just drop the rest, which does NOT cancel them --
275
- // `JoinHandle::drop` detaches, it doesn't abort -- leaving up to
276
- // `MAX_SPLIT_PARTS - 1` sibling Range downloads, each with their
277
- // own `retry_call` backoff, still in flight for a link the caller
278
- // has already given up on).
279
- let mut iter = handles.into_iter();
280
- while let Some(h) = iter.next() {
281
- match h.await {
282
- Ok(Ok(part)) => out.extend_from_slice(&part),
283
- Ok(Err(e)) => {
284
- for remaining in iter {
285
- remaining.abort();
248
+ (index, bytes)
249
+ });
250
+ start = end + 1;
286
251
  }
287
- return Err(e);
288
252
  }
289
- Err(join_err) => {
290
- for remaining in iter {
291
- remaining.abort();
253
+ let head = match resp.bytes().await {
254
+ Ok(bytes) => bytes,
255
+ Err(error) => {
256
+ // Finish cancelling the old parts before retrying
257
+ // the probe under the same download-slot budget.
258
+ downloads.shutdown().await;
259
+ return Err(ApiError::from_reqwest(error, true));
292
260
  }
293
- return Err(join_error(join_err));
294
- }
261
+ };
262
+ Ok((total, head, downloads))
263
+ })
264
+ .await?;
265
+ stats.bytes_downloaded.fetch_add(head.len() as u64, Ordering::Relaxed);
266
+
267
+ let bytes = if downloads.is_empty() {
268
+ head
269
+ } else {
270
+ let mut parts = Vec::with_capacity(downloads.len());
271
+ while let Some(result) = downloads.join_next().await {
272
+ let (index, bytes) = result.map_err(join_error)?;
273
+ parts.push((index, bytes?));
295
274
  }
296
- }
297
- let bytes = out.freeze();
275
+ parts.sort_unstable_by_key(|(index, _)| *index);
276
+ let mut out = bytes::BytesMut::with_capacity(total.unwrap_or(head.len() as u64) as usize);
277
+ out.extend_from_slice(&head);
278
+ for (_, part) in parts {
279
+ out.extend_from_slice(&part);
280
+ }
281
+ out.freeze()
282
+ };
298
283
  if let Some(t) = total
299
284
  && bytes.len() as u64 != t
300
285
  {
@@ -316,6 +301,71 @@ impl DbClient {
316
301
  mod tests {
317
302
  use super::*;
318
303
 
304
+ #[tokio::test]
305
+ async fn split_download_starts_tail_requests_before_the_probe_body_finishes() {
306
+ use tokio::io::{AsyncReadExt, AsyncWriteExt};
307
+ use tokio::net::TcpListener;
308
+
309
+ for fail_first_probe in [false, true] {
310
+ let probes = Arc::new(std::sync::atomic::AtomicUsize::new(0));
311
+ let server_probes = probes.clone();
312
+ let listener = TcpListener::bind("127.0.0.1:0").await.unwrap();
313
+ let address = listener.local_addr().unwrap();
314
+ let tail_started = Arc::new(tokio::sync::Notify::new());
315
+ tokio::spawn(async move {
316
+ for _ in 0..if fail_first_probe { 6 } else { 3 } {
317
+ let (mut socket, _) = listener.accept().await.unwrap();
318
+ let tail_started = tail_started.clone();
319
+ let probes = server_probes.clone();
320
+ tokio::spawn(async move {
321
+ let mut request = Vec::new();
322
+ while !request.ends_with(b"\r\n\r\n") {
323
+ request.push(socket.read_u8().await.unwrap());
324
+ }
325
+ let request = String::from_utf8(request).unwrap().to_ascii_lowercase();
326
+ let range = request.lines().find(|s| s.starts_with("range:")).unwrap();
327
+ let (start, end) = range.split_once("bytes=").unwrap().1.split_once('-').unwrap();
328
+ let start: usize = start.parse().unwrap();
329
+ let end: usize = end.parse().unwrap();
330
+ let headers = format!(
331
+ "HTTP/1.1 206 Partial Content\r\nContent-Length: {}\r\nContent-Range: bytes {start}-{end}/12\r\nConnection: close\r\n\r\n",
332
+ end - start + 1,
333
+ );
334
+ socket.write_all(headers.as_bytes()).await.unwrap();
335
+ if start == 0 {
336
+ let first = probes.fetch_add(1, Ordering::Relaxed) == 0;
337
+ tail_started.notified().await;
338
+ if fail_first_probe && first {
339
+ socket.write_all(b"a").await.unwrap();
340
+ return; // truncated body: the whole probe must retry
341
+ }
342
+ } else {
343
+ tail_started.notify_one();
344
+ }
345
+ let _ = socket.write_all(&b"abcdefghijkl"[start..=end]).await;
346
+ });
347
+ }
348
+ });
349
+ let client = Arc::new(
350
+ DbClient::new(&format!("http://{address}"), "wh", "token")
351
+ .with_retry_attempts(2)
352
+ .with_retry_max_wait_s(0.0),
353
+ );
354
+ let stats = Arc::new(QueryStatsAccumulator::default());
355
+ let bytes = tokio::time::timeout(
356
+ std::time::Duration::from_secs(2),
357
+ client.fetch_link_bytes_split(&format!("http://{address}/blob"), false, 4, 3, &stats),
358
+ )
359
+ .await
360
+ .expect("tail requests must start from the probe headers, without waiting for its body")
361
+ .unwrap();
362
+ assert_eq!(&bytes[..], b"abcdefghijkl");
363
+ assert_eq!(probes.load(Ordering::Relaxed), if fail_first_probe { 2 } else { 1 });
364
+ assert_eq!(stats.retry_count.load(Ordering::Relaxed), u32::from(fail_first_probe));
365
+ assert!(stats.bytes_downloaded.load(Ordering::Relaxed) >= 12);
366
+ }
367
+ }
368
+
319
369
  /// Regression test for a real bug found by testing against an actual
320
370
  /// Databricks workspace (not just synthetic single-frame test data): a
321
371
  /// real chunk's LZ4 compression is several frames concatenated back to
@@ -416,7 +416,7 @@ impl DbClient {
416
416
  // aws-lc-rs to ring, on the theory that either transport change
417
417
  // could have moved the optimum: a first pass reported 96 as
418
418
  // ~13-16% faster than 64 on both this table and a second,
419
- // larger one (fact_sales_order_invoiced, 20M rows/295 cols) --
419
+ // larger one (large_benchmark_table, 20M rows/295 cols) --
420
420
  // **found on review to be a false positive**. The benchmark
421
421
  // script called `connect()`/`arrowbricks.connect()` without
422
422
  // ever passing `chunk_fetch_concurrency=` explicitly, so every
@@ -431,7 +431,7 @@ impl DbClient {
431
431
  // run-to-run noise, not a code effect. Caught by re-running a
432
432
  // controlled, interleaved A/B (`chunk_fetch_concurrency=`
433
433
  // passed explicitly each time, no rebuild needed) on
434
- // `dim_article`: 64/96/64/96 measured 125.60s/129.92s/137.50s/
434
+ // `benchmark_table`: 64/96/64/96 measured 125.60s/129.92s/137.50s/
435
435
  // 134.30s -- no consistent winner, well within run-to-run
436
436
  // noise. Reverted to 64. If re-attempting this again, always
437
437
  // pass `chunk_fetch_concurrency=` explicitly in the benchmark
@@ -17,13 +17,14 @@
17
17
 
18
18
  use std::sync::Arc;
19
19
 
20
- use arrow::array::{
20
+ use arrow_array::RecordBatch;
21
+ use arrow_array::types::Date32Type;
22
+ use arrow_array::{
21
23
  ArrayRef, BinaryArray, BooleanArray, Date32Array, Decimal128Array, Float32Array, Float64Array, Int8Array,
22
24
  Int16Array, Int32Array, Int64Array, StringArray, StructArray, TimestampMicrosecondArray,
23
25
  };
24
- use arrow::buffer::NullBuffer;
25
- use arrow::datatypes::{DataType, Date32Type, Field, Fields, Schema, TimeUnit};
26
- use arrow::record_batch::RecordBatch;
26
+ use arrow_buffer::NullBuffer;
27
+ use arrow_schema::{DataType, Field, Fields, Schema, TimeUnit};
27
28
  use chrono::{DateTime, NaiveDate, NaiveDateTime};
28
29
  use serde_json::Value as JsonValue;
29
30
 
@@ -524,7 +525,7 @@ mod tests {
524
525
  assert_eq!(batch.num_rows(), 1);
525
526
  assert_eq!(batch.num_columns(), 13);
526
527
 
527
- use arrow::array::Array;
528
+ use arrow_array::Array;
528
529
  assert_eq!(
529
530
  batch.column(0).as_any().downcast_ref::<Int8Array>().unwrap().value(0),
530
531
  1
@@ -635,7 +636,7 @@ mod tests {
635
636
  )]];
636
637
 
637
638
  let batch = json_array_to_record_batch(&rows, &columns).unwrap();
638
- use arrow::array::Array;
639
+ use arrow_array::Array;
639
640
  let s = batch.column(0).as_any().downcast_ref::<StructArray>().unwrap();
640
641
  assert_eq!(
641
642
  s.column_by_name("a")
@@ -691,7 +692,7 @@ mod tests {
691
692
  vec![None], // the whole struct is NULL for this row
692
693
  ];
693
694
  let batch = json_array_to_record_batch(&rows, &columns).unwrap();
694
- use arrow::array::Array;
695
+ use arrow_array::Array;
695
696
  let s = batch.column(0).as_any().downcast_ref::<StructArray>().unwrap();
696
697
  assert!(
697
698
  s.column_by_name("a").unwrap().is_null(0),