arrowbricks 3.0.0__tar.gz → 3.0.2__tar.gz
This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
- {arrowbricks-3.0.0 → arrowbricks-3.0.2}/PKG-INFO +1 -1
- {arrowbricks-3.0.0 → arrowbricks-3.0.2}/pyproject.toml +1 -1
- {arrowbricks-3.0.0 → arrowbricks-3.0.2}/rust/arrowbricks_core/Cargo.lock +1 -1
- {arrowbricks-3.0.0 → arrowbricks-3.0.2}/rust/arrowbricks_core/Cargo.toml +15 -3
- {arrowbricks-3.0.0 → arrowbricks-3.0.2}/rust/arrowbricks_core/src/client.rs +180 -168
- {arrowbricks-3.0.0 → arrowbricks-3.0.2}/rust/arrowbricks_core/src/heartbeat.rs +20 -42
- {arrowbricks-3.0.0 → arrowbricks-3.0.2}/rust/arrowbricks_core/src/json_convert.rs +8 -25
- {arrowbricks-3.0.0 → arrowbricks-3.0.2}/rust/arrowbricks_core/src/lib.rs +64 -41
- {arrowbricks-3.0.0 → arrowbricks-3.0.2}/rust/arrowbricks_core/src/pipeline.rs +230 -205
- {arrowbricks-3.0.0 → arrowbricks-3.0.2}/rust/arrowbricks_core/src/thrift.rs +19 -28
- {arrowbricks-3.0.0 → arrowbricks-3.0.2}/rust/arrowbricks_core/tests/wiremock_pipeline.rs +62 -100
- {arrowbricks-3.0.0 → arrowbricks-3.0.2}/rust/arrowbricks_core/tests/wiremock_thrift.rs +141 -2
- {arrowbricks-3.0.0 → arrowbricks-3.0.2}/rust/arrowbricks_core/tests_py/test_token_provider.py +17 -0
- {arrowbricks-3.0.0 → arrowbricks-3.0.2}/rust/arrowbricks_core/tests_py/thrift_mock.py +14 -0
- {arrowbricks-3.0.0 → arrowbricks-3.0.2}/src/arrowbricks/_streaming.py +2 -3
- {arrowbricks-3.0.0 → arrowbricks-3.0.2}/src/arrowbricks/client.py +9 -25
- {arrowbricks-3.0.0 → arrowbricks-3.0.2}/src/arrowbricks/cursor.py +70 -20
- {arrowbricks-3.0.0 → arrowbricks-3.0.2}/LICENSE +0 -0
- {arrowbricks-3.0.0 → arrowbricks-3.0.2}/README.md +0 -0
- {arrowbricks-3.0.0 → arrowbricks-3.0.2}/rust/arrowbricks_core/.gitignore +0 -0
- {arrowbricks-3.0.0 → arrowbricks-3.0.2}/rust/arrowbricks_core/README.md +0 -0
- {arrowbricks-3.0.0 → arrowbricks-3.0.2}/rust/arrowbricks_core/examples/duckdb_query.py +0 -0
- {arrowbricks-3.0.0 → arrowbricks-3.0.2}/rust/arrowbricks_core/examples/fastapi_sse.py +0 -0
- {arrowbricks-3.0.0 → arrowbricks-3.0.2}/rust/arrowbricks_core/rustfmt.toml +0 -0
- {arrowbricks-3.0.0 → arrowbricks-3.0.2}/rust/arrowbricks_core/tests/wiremock_volume_files.rs +0 -0
- {arrowbricks-3.0.0 → arrowbricks-3.0.2}/rust/arrowbricks_core/tests_py/conftest.py +0 -0
- {arrowbricks-3.0.0 → arrowbricks-3.0.2}/rust/arrowbricks_core/tests_py/test_ipc_stream.py +0 -0
- {arrowbricks-3.0.0 → arrowbricks-3.0.2}/rust/arrowbricks_core/tests_py/test_parameters.py +0 -0
- {arrowbricks-3.0.0 → arrowbricks-3.0.2}/rust/arrowbricks_core/tests_py/test_stream_ndjson_lines.py +0 -0
- {arrowbricks-3.0.0 → arrowbricks-3.0.2}/rust/arrowbricks_core/tests_py/test_streaming.py +0 -0
- {arrowbricks-3.0.0 → arrowbricks-3.0.2}/rust/arrowbricks_core/tests_py/test_thrift.py +0 -0
- {arrowbricks-3.0.0 → arrowbricks-3.0.2}/rust/arrowbricks_core/tests_py/test_volume_files.py +0 -0
- {arrowbricks-3.0.0 → arrowbricks-3.0.2}/src/arrowbricks/__init__.py +0 -0
- {arrowbricks-3.0.0 → arrowbricks-3.0.2}/src/arrowbricks/_core.pyi +0 -0
- {arrowbricks-3.0.0 → arrowbricks-3.0.2}/src/arrowbricks/py.typed +0 -0
|
@@ -1,6 +1,6 @@
|
|
|
1
1
|
[project]
|
|
2
2
|
name = "arrowbricks"
|
|
3
|
-
version = "3.0.
|
|
3
|
+
version = "3.0.2"
|
|
4
4
|
description = "Runs SQL against a Databricks SQL warehouse via the Statement Execution API and hands you the result as Arrow -- a DB-API-ish Cursor (fetchone/fetchmany/fetchall/fetchall_arrow) or NDJSON streaming. Rust/PyO3 core throughout -- zero required runtime dependencies."
|
|
5
5
|
readme = "README.md"
|
|
6
6
|
license = "MIT"
|
|
@@ -1,6 +1,6 @@
|
|
|
1
1
|
[package]
|
|
2
2
|
name = "arrowbricks_core"
|
|
3
|
-
version = "3.0.
|
|
3
|
+
version = "3.0.2"
|
|
4
4
|
edition = "2024"
|
|
5
5
|
readme = "README.md"
|
|
6
6
|
|
|
@@ -23,8 +23,13 @@ chrono = { version = "0.4", default-features = false, features = ["std"] }
|
|
|
23
23
|
# Decompresses cloud-fetch chunk bytes when the server honors our
|
|
24
24
|
# `result_compression: LZ4_FRAME` request (see client.rs's execute_statement)
|
|
25
25
|
# -- the same LZ4 Frame format databricks-sql-python's `lz4.frame` decodes.
|
|
26
|
-
# Pure Rust
|
|
27
|
-
#
|
|
26
|
+
# Pure Rust, no C toolchain dependency -- unlike reqwest's `rustls` feature
|
|
27
|
+
# below, which *does* pull one in (`__rustls-aws-lc-rs` -> aws-lc-rs ->
|
|
28
|
+
# aws-lc-sys, confirmed via `cargo tree -i aws-lc-sys`; this crate has no
|
|
29
|
+
# `ring` feature available in reqwest 0.13 to avoid it with). A previous
|
|
30
|
+
# version of this comment claimed the opposite -- that this crate avoids
|
|
31
|
+
# aws-lc-sys-style native-build pain elsewhere too -- found wrong on review;
|
|
32
|
+
# see AGENTS.md's build-prerequisites note.
|
|
28
33
|
lz4_flex = { version = "0.11", default-features = false, features = ["frame", "std"] }
|
|
29
34
|
pyo3 = { version = "0.29.1", features = ["abi3-py311"] }
|
|
30
35
|
# pyo3-arrow's only optional feature is buffer_protocol (on by default),
|
|
@@ -53,6 +58,13 @@ extension-module = ["pyo3/extension-module"]
|
|
|
53
58
|
tokio = { version = "1.53.1", features = ["rt-multi-thread", "macros", "test-util", "net", "io-util"] }
|
|
54
59
|
wiremock = "0.6.5"
|
|
55
60
|
|
|
61
|
+
# Currently clean on all three by luck, not enforcement -- pinning them here
|
|
62
|
+
# so a future PR that introduces one fails CI instead of silently landing.
|
|
63
|
+
[lints.clippy]
|
|
64
|
+
redundant_clone = "warn"
|
|
65
|
+
large_enum_variant = "warn"
|
|
66
|
+
needless_collect = "warn"
|
|
67
|
+
|
|
56
68
|
[profile.release]
|
|
57
69
|
# opt-level stays at the implicit default (3, speed) rather than "s"/"z" --
|
|
58
70
|
# perf was the original ask and the benchmarks were run against opt-level 3;
|
|
@@ -148,8 +148,7 @@ struct ResultBody {
|
|
|
148
148
|
external_links: Vec<ResultLinkBody>,
|
|
149
149
|
/// Only present for `disposition: INLINE` + `format: JSON_ARRAY` -- each
|
|
150
150
|
/// row a `Vec<Option<String>>` (every non-null value a string,
|
|
151
|
-
/// Databricks' own JSON_ARRAY contract
|
|
152
|
-
/// normal `EXTERNAL_LINKS`+`JSON_ARRAY` chunks). Consumed by
|
|
151
|
+
/// Databricks' own JSON_ARRAY contract). Consumed by
|
|
153
152
|
/// `json_convert::json_array_to_record_batch` on the `prefer_inline`
|
|
154
153
|
/// fast path.
|
|
155
154
|
#[serde(default)]
|
|
@@ -329,7 +328,7 @@ pub struct StatementSubmitResult {
|
|
|
329
328
|
|
|
330
329
|
/// What `execute_arrow_statement_prefer_inline` gets you -- either the whole
|
|
331
330
|
/// result inline as raw JSON_ARRAY rows (every non-null value a string,
|
|
332
|
-
///
|
|
331
|
+
/// Databricks' own JSON_ARRAY contract; converting these into a real
|
|
333
332
|
/// `RecordBatch` is `pipeline.rs`'s job via `json_convert`, not this
|
|
334
333
|
/// Arrow-agnostic module's -- see this file's own module doc comment), or a
|
|
335
334
|
/// normal `StatementSubmitResult` to fetch chunks for exactly as if
|
|
@@ -353,15 +352,18 @@ pub struct ChunkItem {
|
|
|
353
352
|
pub blob: Bytes,
|
|
354
353
|
pub row_count: Option<i64>,
|
|
355
354
|
pub chunk_index: i64,
|
|
356
|
-
///
|
|
357
|
-
///
|
|
358
|
-
/// decoded batches must be sliced down to if they exceed it, since a
|
|
359
|
-
/// Thrift `resultLinks` file (like SEA's own cloud-fetch files) can
|
|
355
|
+
/// A declared row-count bound this chunk's own decoded batches must be
|
|
356
|
+
/// sliced down to if they exceed it, since a cloud-fetch file can
|
|
360
357
|
/// legitimately contain more rows than its own declared count for a
|
|
361
358
|
/// `LIMIT`-bounded query (see `pipeline.rs`'s `decode_chunk_item`).
|
|
362
|
-
///
|
|
363
|
-
///
|
|
364
|
-
///
|
|
359
|
+
/// Every real producer sets this now -- Thrift `resultLinks`
|
|
360
|
+
/// (`pipeline.rs`'s `fetch_thrift_link`), Thrift's inline `arrowBatches`
|
|
361
|
+
/// (`pipeline.rs`'s `run_thrift_fetch_loop`), and SEA's own chunk fetch
|
|
362
|
+
/// (`fetch_chunks_with_backpressure`, below, though only when a
|
|
363
|
+
/// `chunk_index` resolves to exactly one blob -- see that function's own
|
|
364
|
+
/// comment for why more than one blob is left untruncated). `None`
|
|
365
|
+
/// means no authoritative bound is known, not "this chunk was never
|
|
366
|
+
/// overshot" -- `decode_chunk_item` treats it as a no-op either way.
|
|
365
367
|
pub truncate_to: Option<i64>,
|
|
366
368
|
}
|
|
367
369
|
|
|
@@ -507,9 +509,21 @@ pub struct DbClient {
|
|
|
507
509
|
/// rebuild the extension to do it. Also doubles, on the Thrift path, as
|
|
508
510
|
/// `TExecuteStatementReq.canDecompressLZ4Result`.
|
|
509
511
|
compress_results: bool,
|
|
510
|
-
session_pool:
|
|
512
|
+
session_pool: Pool<String>,
|
|
511
513
|
pub protocol: Protocol,
|
|
512
|
-
|
|
514
|
+
/// Same checkout/checkin shape as `session_pool` above, plus: Thrift's
|
|
515
|
+
/// session model is *mandatory* per statement, unlike SEA's optional
|
|
516
|
+
/// `session_id`, and this crate has not independently confirmed whether
|
|
517
|
+
/// concurrent statements on one Thrift session are safe against a real
|
|
518
|
+
/// workspace the way the SEA crash was -- so this pool exists
|
|
519
|
+
/// defensively, following the same proven-safe pattern regardless of
|
|
520
|
+
/// whether the analogous crash reproduces here. Unlike SEA, there is no
|
|
521
|
+
/// "sessionless" fallback available at all -- `TExecuteStatementReq.
|
|
522
|
+
/// sessionHandle` is required by the protocol -- so a pool-exhaustion/
|
|
523
|
+
/// creation-failure `None` from `thrift_checkout_session` means the
|
|
524
|
+
/// caller must open one throwaway session for that single call and
|
|
525
|
+
/// close it again immediately after (see `execute_lazy_thrift`).
|
|
526
|
+
thrift_session_pool: Pool<thrift::SessionHandle>,
|
|
513
527
|
/// Global budget for concurrent cloud-fetch HTTP requests, sized to
|
|
514
528
|
/// `chunk_fetch_concurrency`. Every link download takes one permit; a
|
|
515
529
|
/// download that finds spare permits (few links in flight -- exactly
|
|
@@ -529,24 +543,12 @@ pub struct DbClient {
|
|
|
529
543
|
/// ever split into -- see `DbClient::download_slots`.
|
|
530
544
|
pub const MAX_SPLIT_PARTS: usize = 8;
|
|
531
545
|
|
|
532
|
-
///
|
|
533
|
-
///
|
|
534
|
-
///
|
|
535
|
-
///
|
|
536
|
-
///
|
|
537
|
-
|
|
538
|
-
/// workspace the way the SEA crash was -- so this pool exists defensively,
|
|
539
|
-
/// following the same proven-safe pattern regardless of whether the
|
|
540
|
-
/// analogous crash reproduces here. Unlike SEA, there is no "sessionless"
|
|
541
|
-
/// fallback available at all -- `TExecuteStatementReq.sessionHandle` is
|
|
542
|
-
/// required by the protocol -- so a pool-exhaustion/creation-failure `None`
|
|
543
|
-
/// here means the caller must open one throwaway session for that single
|
|
544
|
-
/// call and close it again immediately after (see `execute_lazy_thrift`).
|
|
545
|
-
#[derive(Default)]
|
|
546
|
-
struct ThriftSessionPool {
|
|
547
|
-
idle: Mutex<HashMap<(Option<String>, Option<String>), Vec<thrift::SessionHandle>>>,
|
|
548
|
-
total: Mutex<HashMap<(Option<String>, Option<String>), usize>>,
|
|
549
|
-
}
|
|
546
|
+
/// `chunk_fetch_concurrency`'s own default -- see `DbClient::new`'s own
|
|
547
|
+
/// comment on it for the measured numbers behind picking 64. Named so
|
|
548
|
+
/// `download_slots` (sized to this same number, since it's a budget over
|
|
549
|
+
/// the same worker count) can't drift from it the way a second bare `64`
|
|
550
|
+
/// silently could.
|
|
551
|
+
const DEFAULT_CHUNK_FETCH_CONCURRENCY: usize = 64;
|
|
550
552
|
|
|
551
553
|
/// How long the Thrift path polls `GetOperationStatus` when a statement
|
|
552
554
|
/// doesn't finish within its `getDirectResults` budget (see
|
|
@@ -621,13 +623,77 @@ const THRIFT_DIRECT_RESULTS_MAX_BYTES: i64 = 1024 * 1024 * 1024;
|
|
|
621
623
|
/// still throws away a perfectly good session), but guarantees a session
|
|
622
624
|
/// that might be in the same bad state behind the SparkSession-null crash
|
|
623
625
|
/// above is never handed to a second caller.
|
|
624
|
-
|
|
625
|
-
|
|
626
|
+
type SessionKey = (Option<String>, Option<String>);
|
|
627
|
+
|
|
628
|
+
/// Generic checkout/checkin pool keyed by (catalog, schema) -- shared by
|
|
629
|
+
/// SEA's session ids (`Pool<String>`) and Thrift's session handles
|
|
630
|
+
/// (`Pool<thrift::SessionHandle>`), which had this exact logic duplicated
|
|
631
|
+
/// twice over before this. See `checkout_session`/`thrift_checkout_session`
|
|
632
|
+
/// for the constraints that shape it (session pinned to one (catalog,
|
|
633
|
+
/// schema) pair, safe for sequential reuse only, checkout never blocks on
|
|
634
|
+
/// exhaustion).
|
|
635
|
+
struct Pool<T> {
|
|
626
636
|
// ponytail: fixed cap, not a constructor kwarg -- nothing's asked to
|
|
627
637
|
// tune this yet; raise (or expose one) if a workload needs more
|
|
628
638
|
// concurrent sessions per (catalog, schema) pair than this.
|
|
629
|
-
idle: Mutex<HashMap<
|
|
630
|
-
total: Mutex<HashMap<
|
|
639
|
+
idle: Mutex<HashMap<SessionKey, Vec<T>>>,
|
|
640
|
+
total: Mutex<HashMap<SessionKey, usize>>,
|
|
641
|
+
}
|
|
642
|
+
|
|
643
|
+
// Manual impl instead of `#[derive(Default)]`: the derive would wrongly
|
|
644
|
+
// require `T: Default` even though neither field actually needs it.
|
|
645
|
+
impl<T> Default for Pool<T> {
|
|
646
|
+
fn default() -> Self {
|
|
647
|
+
Self {
|
|
648
|
+
idle: Mutex::new(HashMap::new()),
|
|
649
|
+
total: Mutex::new(HashMap::new()),
|
|
650
|
+
}
|
|
651
|
+
}
|
|
652
|
+
}
|
|
653
|
+
|
|
654
|
+
impl<T> Pool<T> {
|
|
655
|
+
fn take(&self, key: &SessionKey) -> Option<T> {
|
|
656
|
+
self.idle.lock().unwrap().get_mut(key).and_then(Vec::pop)
|
|
657
|
+
}
|
|
658
|
+
|
|
659
|
+
/// Reserves a slot for `key` if under `MAX_SESSIONS_PER_KEY`. The
|
|
660
|
+
/// increment happens before the caller's own creation call so two
|
|
661
|
+
/// concurrent callers can't both squeeze past the cap; call `release` if
|
|
662
|
+
/// creation then fails.
|
|
663
|
+
fn reserve(&self, key: &SessionKey) -> bool {
|
|
664
|
+
let mut total = self.total.lock().unwrap();
|
|
665
|
+
let count = total.entry(key.clone()).or_insert(0);
|
|
666
|
+
if *count >= MAX_SESSIONS_PER_KEY {
|
|
667
|
+
return false;
|
|
668
|
+
}
|
|
669
|
+
*count += 1;
|
|
670
|
+
true
|
|
671
|
+
}
|
|
672
|
+
|
|
673
|
+
fn release(&self, key: &SessionKey) {
|
|
674
|
+
let mut total = self.total.lock().unwrap();
|
|
675
|
+
if let Some(count) = total.get_mut(key) {
|
|
676
|
+
*count = count.saturating_sub(1);
|
|
677
|
+
}
|
|
678
|
+
}
|
|
679
|
+
|
|
680
|
+
fn put(&self, key: SessionKey, item: T) {
|
|
681
|
+
self.idle.lock().unwrap().entry(key).or_default().push(item);
|
|
682
|
+
}
|
|
683
|
+
|
|
684
|
+
/// Returns `item` to the pool for reuse (`keep = true`) or discards it,
|
|
685
|
+
/// releasing its reservation instead (`keep = false`).
|
|
686
|
+
fn checkin(&self, key: SessionKey, item: T, keep: bool) {
|
|
687
|
+
if keep {
|
|
688
|
+
self.put(key, item);
|
|
689
|
+
} else {
|
|
690
|
+
self.release(&key);
|
|
691
|
+
}
|
|
692
|
+
}
|
|
693
|
+
|
|
694
|
+
fn drain_idle(&self) -> Vec<T> {
|
|
695
|
+
self.idle.lock().unwrap().drain().flat_map(|(_, v)| v).collect()
|
|
696
|
+
}
|
|
631
697
|
}
|
|
632
698
|
|
|
633
699
|
pub const MAX_SESSIONS_PER_KEY: usize = 8;
|
|
@@ -672,15 +738,15 @@ impl DbClient {
|
|
|
672
738
|
// a claim that 64 is the true optimum. A caller with a very
|
|
673
739
|
// large chunk count or a fast/low-latency link to the warehouse
|
|
674
740
|
// may still want to raise it further.
|
|
675
|
-
chunk_fetch_concurrency:
|
|
741
|
+
chunk_fetch_concurrency: DEFAULT_CHUNK_FETCH_CONCURRENCY,
|
|
676
742
|
warehouse_start_timeout: Duration::from_secs(300),
|
|
677
743
|
warehouse_confirmed_running_ttl: Duration::from_secs(30),
|
|
678
744
|
warehouse_confirmed_running_at: Mutex::new(None),
|
|
679
745
|
compress_results: true,
|
|
680
|
-
session_pool:
|
|
746
|
+
session_pool: Pool::default(),
|
|
681
747
|
protocol: Protocol::Thrift,
|
|
682
|
-
thrift_session_pool:
|
|
683
|
-
download_slots: tokio::sync::Semaphore::new(
|
|
748
|
+
thrift_session_pool: Pool::default(),
|
|
749
|
+
download_slots: tokio::sync::Semaphore::new(DEFAULT_CHUNK_FETCH_CONCURRENCY),
|
|
684
750
|
}
|
|
685
751
|
}
|
|
686
752
|
|
|
@@ -690,11 +756,7 @@ impl DbClient {
|
|
|
690
756
|
self
|
|
691
757
|
}
|
|
692
758
|
|
|
693
|
-
/// Selects the wire protocol/backend -- `Protocol
|
|
694
|
-
/// `DbClient` constructor's own internal starting value, see
|
|
695
|
-
/// `Protocol`'s own doc comment for why that's not the same thing as
|
|
696
|
-
/// "the user-facing default") or `Protocol::Thrift` (the actual
|
|
697
|
-
/// user-facing default as of this session's benchmarking work).
|
|
759
|
+
/// Selects the wire protocol/backend -- see `Protocol`'s own doc comment.
|
|
698
760
|
pub fn with_protocol(mut self, protocol: Protocol) -> Self {
|
|
699
761
|
self.protocol = protocol;
|
|
700
762
|
self
|
|
@@ -830,11 +892,9 @@ impl DbClient {
|
|
|
830
892
|
}
|
|
831
893
|
|
|
832
894
|
/// Downloads one cloud-fetch link, splitting it across parallel HTTP
|
|
833
|
-
/// Range requests when
|
|
834
|
-
///
|
|
835
|
-
///
|
|
836
|
-
/// stream leaves most of the link idle. See `download_slots`' own doc
|
|
837
|
-
/// comment for the measured win and why the large-result case is safe.
|
|
895
|
+
/// Range requests when the shared `download_slots` budget has room to
|
|
896
|
+
/// spare -- see that field's own doc comment for the measured win and
|
|
897
|
+
/// why the large-result case is safe.
|
|
838
898
|
pub(crate) async fn fetch_link_bytes_budgeted(
|
|
839
899
|
self: &Arc<Self>,
|
|
840
900
|
url: &str,
|
|
@@ -843,8 +903,16 @@ impl DbClient {
|
|
|
843
903
|
// One permit per download is mandatory; the worker pool already
|
|
844
904
|
// bounds concurrent links to `chunk_fetch_concurrency`, so this
|
|
845
905
|
// never blocks in practice -- it just makes the budget accounting
|
|
846
|
-
// exact.
|
|
847
|
-
|
|
906
|
+
// exact. `acquire()`'s `Err` case (the semaphore closed) can't
|
|
907
|
+
// happen -- nothing in this crate ever calls `.close()` on
|
|
908
|
+
// `download_slots` -- but propagating it as a real error instead of
|
|
909
|
+
// an `.expect()` costs nothing and avoids a panic if that ever
|
|
910
|
+
// changes.
|
|
911
|
+
let _base = self
|
|
912
|
+
.download_slots
|
|
913
|
+
.acquire()
|
|
914
|
+
.await
|
|
915
|
+
.map_err(|e| ApiError::permanent(format!("download_slots semaphore closed unexpectedly: {e}")))?;
|
|
848
916
|
let want = (MAX_SPLIT_PARTS - 1).min(self.download_slots.available_permits());
|
|
849
917
|
let extra = if want > 0 {
|
|
850
918
|
self.download_slots.try_acquire_many(want as u32).ok()
|
|
@@ -960,8 +1028,33 @@ impl DbClient {
|
|
|
960
1028
|
|
|
961
1029
|
let mut out = bytes::BytesMut::with_capacity(total.unwrap_or(head.len() as u64) as usize);
|
|
962
1030
|
out.extend_from_slice(&head);
|
|
963
|
-
|
|
964
|
-
|
|
1031
|
+
// Parts must concatenate in order, so this can't use `JoinSet`
|
|
1032
|
+
// (which yields in completion order) without tracking indices --
|
|
1033
|
+
// simpler to keep the `Vec` and explicitly `.abort()` every
|
|
1034
|
+
// not-yet-awaited sibling the moment one part fails, rather than
|
|
1035
|
+
// silently leaving them running (a bare `?` here would return
|
|
1036
|
+
// early and just drop the rest, which does NOT cancel them --
|
|
1037
|
+
// `JoinHandle::drop` detaches, it doesn't abort -- leaving up to
|
|
1038
|
+
// `MAX_SPLIT_PARTS - 1` sibling Range downloads, each with their
|
|
1039
|
+
// own `retry_call` backoff, still in flight for a link the caller
|
|
1040
|
+
// has already given up on).
|
|
1041
|
+
let mut iter = handles.into_iter();
|
|
1042
|
+
while let Some(h) = iter.next() {
|
|
1043
|
+
match h.await {
|
|
1044
|
+
Ok(Ok(part)) => out.extend_from_slice(&part),
|
|
1045
|
+
Ok(Err(e)) => {
|
|
1046
|
+
for remaining in iter {
|
|
1047
|
+
remaining.abort();
|
|
1048
|
+
}
|
|
1049
|
+
return Err(e);
|
|
1050
|
+
}
|
|
1051
|
+
Err(join_err) => {
|
|
1052
|
+
for remaining in iter {
|
|
1053
|
+
remaining.abort();
|
|
1054
|
+
}
|
|
1055
|
+
return Err(join_error(join_err));
|
|
1056
|
+
}
|
|
1057
|
+
}
|
|
965
1058
|
}
|
|
966
1059
|
let bytes = out.freeze();
|
|
967
1060
|
if let Some(t) = total
|
|
@@ -980,13 +1073,16 @@ impl DbClient {
|
|
|
980
1073
|
.map_err(join_error)?
|
|
981
1074
|
}
|
|
982
1075
|
|
|
983
|
-
|
|
1076
|
+
/// Shared by both protocols -- plain REST against `/api/2.0/sql/warehouses/{id}`,
|
|
1077
|
+
/// nothing SEA- or Thrift-specific about it (see AGENTS.md for why the
|
|
1078
|
+
/// Thrift path didn't always call this).
|
|
1079
|
+
pub(crate) async fn ensure_warehouse_running(&self) -> Result<(), ApiError> {
|
|
984
1080
|
{
|
|
985
1081
|
let confirmed = *self.warehouse_confirmed_running_at.lock().unwrap();
|
|
986
|
-
if let Some(at) = confirmed
|
|
987
|
-
|
|
988
|
-
|
|
989
|
-
|
|
1082
|
+
if let Some(at) = confirmed
|
|
1083
|
+
&& at.elapsed() < self.warehouse_confirmed_running_ttl
|
|
1084
|
+
{
|
|
1085
|
+
return Ok(());
|
|
990
1086
|
}
|
|
991
1087
|
}
|
|
992
1088
|
|
|
@@ -1041,36 +1137,20 @@ impl DbClient {
|
|
|
1041
1137
|
/// Hands back a pooled session for (`catalog`, `schema`) if one's idle,
|
|
1042
1138
|
/// creates one if the pool for that key isn't at `MAX_SESSIONS_PER_KEY`
|
|
1043
1139
|
/// yet, or `None` if neither -- the caller falls back to a plain
|
|
1044
|
-
/// session-less submission in that case, see `
|
|
1045
|
-
/// comment for why this never blocks instead.
|
|
1046
|
-
/// increment happens *before* the `create_session` await so two
|
|
1047
|
-
/// concurrent callers can't both squeeze past the cap; a failed creation
|
|
1048
|
-
/// releases the reservation again.
|
|
1140
|
+
/// session-less submission in that case, see `session_pool`'s own doc
|
|
1141
|
+
/// comment for why this never blocks instead.
|
|
1049
1142
|
async fn checkout_session(&self, catalog: Option<&str>, schema: Option<&str>) -> Option<String> {
|
|
1050
1143
|
let key = (catalog.map(str::to_string), schema.map(str::to_string));
|
|
1051
|
-
{
|
|
1052
|
-
|
|
1053
|
-
if let Some(ids) = idle.get_mut(&key)
|
|
1054
|
-
&& let Some(id) = ids.pop()
|
|
1055
|
-
{
|
|
1056
|
-
return Some(id);
|
|
1057
|
-
}
|
|
1144
|
+
if let Some(id) = self.session_pool.take(&key) {
|
|
1145
|
+
return Some(id);
|
|
1058
1146
|
}
|
|
1059
|
-
{
|
|
1060
|
-
|
|
1061
|
-
let count = total.entry(key.clone()).or_insert(0);
|
|
1062
|
-
if *count >= MAX_SESSIONS_PER_KEY {
|
|
1063
|
-
return None;
|
|
1064
|
-
}
|
|
1065
|
-
*count += 1;
|
|
1147
|
+
if !self.session_pool.reserve(&key) {
|
|
1148
|
+
return None;
|
|
1066
1149
|
}
|
|
1067
1150
|
match self.create_session(catalog, schema).await {
|
|
1068
1151
|
Ok(id) => Some(id),
|
|
1069
1152
|
Err(_) => {
|
|
1070
|
-
|
|
1071
|
-
if let Some(count) = total.get_mut(&key) {
|
|
1072
|
-
*count = count.saturating_sub(1);
|
|
1073
|
-
}
|
|
1153
|
+
self.session_pool.release(&key);
|
|
1074
1154
|
None
|
|
1075
1155
|
}
|
|
1076
1156
|
}
|
|
@@ -1078,24 +1158,11 @@ impl DbClient {
|
|
|
1078
1158
|
|
|
1079
1159
|
/// Returns a session to the pool for reuse (`keep = true`, the statement
|
|
1080
1160
|
/// it backed reached SUCCEEDED/FAILED/CANCELED cleanly) or discards it
|
|
1081
|
-
/// (`keep = false`) -- see `
|
|
1161
|
+
/// (`keep = false`) -- see `session_pool`'s doc comment for why any error
|
|
1082
1162
|
/// discards rather than reuses.
|
|
1083
1163
|
fn checkin_session(&self, catalog: Option<&str>, schema: Option<&str>, session_id: String, keep: bool) {
|
|
1084
1164
|
let key = (catalog.map(str::to_string), schema.map(str::to_string));
|
|
1085
|
-
|
|
1086
|
-
self.session_pool
|
|
1087
|
-
.idle
|
|
1088
|
-
.lock()
|
|
1089
|
-
.unwrap()
|
|
1090
|
-
.entry(key)
|
|
1091
|
-
.or_default()
|
|
1092
|
-
.push(session_id);
|
|
1093
|
-
} else {
|
|
1094
|
-
let mut total = self.session_pool.total.lock().unwrap();
|
|
1095
|
-
if let Some(count) = total.get_mut(&key) {
|
|
1096
|
-
*count = count.saturating_sub(1);
|
|
1097
|
-
}
|
|
1098
|
-
}
|
|
1165
|
+
self.session_pool.checkin(key, session_id, keep);
|
|
1099
1166
|
}
|
|
1100
1167
|
|
|
1101
1168
|
/// Best-effort cleanup of every currently-idle pooled session -- meant
|
|
@@ -1106,11 +1173,7 @@ impl DbClient {
|
|
|
1106
1173
|
/// this before every pending statement has finished is a caller
|
|
1107
1174
|
/// ordering issue, not something this method can fix from inside.
|
|
1108
1175
|
pub async fn close_all_sessions(&self) {
|
|
1109
|
-
|
|
1110
|
-
let mut idle = self.session_pool.idle.lock().unwrap();
|
|
1111
|
-
idle.drain().flat_map(|(_, v)| v).collect()
|
|
1112
|
-
};
|
|
1113
|
-
for id in ids {
|
|
1176
|
+
for id in self.session_pool.drain_idle() {
|
|
1114
1177
|
self.delete_session(&id).await;
|
|
1115
1178
|
}
|
|
1116
1179
|
}
|
|
@@ -1201,8 +1264,8 @@ impl DbClient {
|
|
|
1201
1264
|
}
|
|
1202
1265
|
|
|
1203
1266
|
/// Same checkout contract as SEA's `checkout_session` (see
|
|
1204
|
-
/// `
|
|
1205
|
-
/// its own throwaway session for this one call (Thrift has no
|
|
1267
|
+
/// `thrift_session_pool`'s doc comment): `None` means the caller must
|
|
1268
|
+
/// open its own throwaway session for this one call (Thrift has no
|
|
1206
1269
|
/// session-less submission mode to fall back to).
|
|
1207
1270
|
pub(crate) async fn thrift_checkout_session(
|
|
1208
1271
|
&self,
|
|
@@ -1210,29 +1273,16 @@ impl DbClient {
|
|
|
1210
1273
|
schema: Option<&str>,
|
|
1211
1274
|
) -> Option<thrift::SessionHandle> {
|
|
1212
1275
|
let key = (catalog.map(str::to_string), schema.map(str::to_string));
|
|
1213
|
-
{
|
|
1214
|
-
|
|
1215
|
-
if let Some(ids) = idle.get_mut(&key)
|
|
1216
|
-
&& let Some(id) = ids.pop()
|
|
1217
|
-
{
|
|
1218
|
-
return Some(id);
|
|
1219
|
-
}
|
|
1276
|
+
if let Some(handle) = self.thrift_session_pool.take(&key) {
|
|
1277
|
+
return Some(handle);
|
|
1220
1278
|
}
|
|
1221
|
-
{
|
|
1222
|
-
|
|
1223
|
-
let count = total.entry(key.clone()).or_insert(0);
|
|
1224
|
-
if *count >= MAX_SESSIONS_PER_KEY {
|
|
1225
|
-
return None;
|
|
1226
|
-
}
|
|
1227
|
-
*count += 1;
|
|
1279
|
+
if !self.thrift_session_pool.reserve(&key) {
|
|
1280
|
+
return None;
|
|
1228
1281
|
}
|
|
1229
1282
|
match self.thrift_open_session_raw(catalog, schema).await {
|
|
1230
1283
|
Ok(handle) => Some(handle),
|
|
1231
1284
|
Err(_) => {
|
|
1232
|
-
|
|
1233
|
-
if let Some(count) = total.get_mut(&key) {
|
|
1234
|
-
*count = count.saturating_sub(1);
|
|
1235
|
-
}
|
|
1285
|
+
self.thrift_session_pool.release(&key);
|
|
1236
1286
|
None
|
|
1237
1287
|
}
|
|
1238
1288
|
}
|
|
@@ -1246,30 +1296,13 @@ impl DbClient {
|
|
|
1246
1296
|
keep: bool,
|
|
1247
1297
|
) {
|
|
1248
1298
|
let key = (catalog.map(str::to_string), schema.map(str::to_string));
|
|
1249
|
-
|
|
1250
|
-
self.thrift_session_pool
|
|
1251
|
-
.idle
|
|
1252
|
-
.lock()
|
|
1253
|
-
.unwrap()
|
|
1254
|
-
.entry(key)
|
|
1255
|
-
.or_default()
|
|
1256
|
-
.push(session);
|
|
1257
|
-
} else {
|
|
1258
|
-
let mut total = self.thrift_session_pool.total.lock().unwrap();
|
|
1259
|
-
if let Some(count) = total.get_mut(&key) {
|
|
1260
|
-
*count = count.saturating_sub(1);
|
|
1261
|
-
}
|
|
1262
|
-
}
|
|
1299
|
+
self.thrift_session_pool.checkin(key, session, keep);
|
|
1263
1300
|
}
|
|
1264
1301
|
|
|
1265
1302
|
/// Best-effort close of every currently-idle pooled Thrift session --
|
|
1266
1303
|
/// same contract as `close_all_sessions` (SEA).
|
|
1267
1304
|
pub async fn close_all_thrift_sessions(&self) {
|
|
1268
|
-
|
|
1269
|
-
let mut idle = self.thrift_session_pool.idle.lock().unwrap();
|
|
1270
|
-
idle.drain().flat_map(|(_, v)| v).collect()
|
|
1271
|
-
};
|
|
1272
|
-
for s in sessions {
|
|
1305
|
+
for s in self.thrift_session_pool.drain_idle() {
|
|
1273
1306
|
self.thrift_close_session_raw(&s).await;
|
|
1274
1307
|
}
|
|
1275
1308
|
}
|
|
@@ -1305,7 +1338,6 @@ impl DbClient {
|
|
|
1305
1338
|
op: &thrift::OperationHandle,
|
|
1306
1339
|
) -> Result<thrift::OperationStatusResp, ApiError> {
|
|
1307
1340
|
let body = Bytes::from(thrift::build_get_operation_status(op));
|
|
1308
|
-
// Idempotent -- read-only.
|
|
1309
1341
|
let resp_bytes = self.thrift_call(body, true).await?;
|
|
1310
1342
|
thrift::parse_get_operation_status(&resp_bytes).map_err(Self::thrift_parse_error)
|
|
1311
1343
|
}
|
|
@@ -1438,24 +1470,6 @@ impl DbClient {
|
|
|
1438
1470
|
}
|
|
1439
1471
|
}
|
|
1440
1472
|
|
|
1441
|
-
/// Like `execute_arrow_statement`, fixed to JSON_ARRAY -- each fetched
|
|
1442
|
-
/// chunk's bytes are then a JSON array of rows, each row itself an array
|
|
1443
|
-
/// of values where every non-null value is a *string* regardless of its
|
|
1444
|
-
/// real column type (Databricks' own JSON_ARRAY contract; null stays
|
|
1445
|
-
/// JSON null) -- casting by the manifest's column type_name, if wanted,
|
|
1446
|
-
/// is left to the caller, same as the Python original does nothing extra
|
|
1447
|
-
/// here either.
|
|
1448
|
-
pub async fn execute_json_statement(
|
|
1449
|
-
&self,
|
|
1450
|
-
statement: &str,
|
|
1451
|
-
catalog: Option<&str>,
|
|
1452
|
-
schema: Option<&str>,
|
|
1453
|
-
parameters: Option<Value>,
|
|
1454
|
-
) -> Result<StatementSubmitResult, ApiError> {
|
|
1455
|
-
self.execute_statement(statement, "JSON_ARRAY", catalog, schema, parameters)
|
|
1456
|
-
.await
|
|
1457
|
-
}
|
|
1458
|
-
|
|
1459
1473
|
/// Shared by `execute_statement` and `execute_arrow_statement_prefer_inline`:
|
|
1460
1474
|
/// POST the statement, poll until a terminal state, and turn FAILED/
|
|
1461
1475
|
/// CANCELED into an `Err` -- everything both callers need before they
|
|
@@ -1463,7 +1477,7 @@ impl DbClient {
|
|
|
1463
1477
|
///
|
|
1464
1478
|
/// `body` must *not* already carry `catalog`/`schema`/`session_id` --
|
|
1465
1479
|
/// this method owns that decision: a pooled session for (`catalog`,
|
|
1466
|
-
/// `schema`) if one's available (see `checkout_session
|
|
1480
|
+
/// `schema`) if one's available (see `checkout_session`),
|
|
1467
1481
|
/// falling back to setting `catalog`/`schema` directly on the body
|
|
1468
1482
|
/// otherwise (Databricks rejects `session_id` combined with either
|
|
1469
1483
|
/// field). The session, if any, is returned to the pool on a clean
|
|
@@ -1601,12 +1615,8 @@ impl DbClient {
|
|
|
1601
1615
|
self.host, statement_id, chunk_index
|
|
1602
1616
|
);
|
|
1603
1617
|
let data: ChunkLinksBody = self.authed_json(reqwest::Method::GET, &url, None).await?;
|
|
1604
|
-
|
|
1605
|
-
|
|
1606
|
-
for link in data.external_links {
|
|
1607
|
-
blobs.push(self.fetch_link_bytes(&link.external_link, compressed).await?);
|
|
1608
|
-
}
|
|
1609
|
-
Ok(blobs)
|
|
1618
|
+
let links: Vec<String> = data.external_links.into_iter().map(|l| l.external_link).collect();
|
|
1619
|
+
self.fetch_pre_resolved_links(&links, compressed).await
|
|
1610
1620
|
}
|
|
1611
1621
|
|
|
1612
1622
|
/// Same shape as `fetch_chunk_index`, but for links the statement submit/
|
|
@@ -1661,12 +1671,16 @@ impl DbClient {
|
|
|
1661
1671
|
};
|
|
1662
1672
|
match fetched {
|
|
1663
1673
|
Ok(blobs) => {
|
|
1674
|
+
// One blob ⇒ `meta.row_count` (the whole chunk_index's
|
|
1675
|
+
// declared count) and "this blob's count" are the same
|
|
1676
|
+
// number -- see `ChunkItem::truncate_to`'s own doc comment.
|
|
1677
|
+
let truncate_to = if blobs.len() == 1 { meta.row_count } else { None };
|
|
1664
1678
|
for blob in blobs {
|
|
1665
1679
|
let item = ChunkItem {
|
|
1666
1680
|
blob,
|
|
1667
1681
|
row_count: meta.row_count,
|
|
1668
1682
|
chunk_index: meta.chunk_index,
|
|
1669
|
-
truncate_to
|
|
1683
|
+
truncate_to,
|
|
1670
1684
|
};
|
|
1671
1685
|
if worker_tx.send(Ok(item)).await.is_err() {
|
|
1672
1686
|
return Ok(());
|
|
@@ -1737,8 +1751,6 @@ impl DbClient {
|
|
|
1737
1751
|
.timeout(self.http_timeout)
|
|
1738
1752
|
.send()
|
|
1739
1753
|
.await
|
|
1740
|
-
// DELETE is naturally idempotent here too (404 is already
|
|
1741
|
-
// treated as success below).
|
|
1742
1754
|
.map_err(|e| ApiError::from_reqwest(e, true))?;
|
|
1743
1755
|
let status = resp.status();
|
|
1744
1756
|
if status == StatusCode::NOT_FOUND || status.is_success() {
|