arrowbricks 3.0.1__tar.gz → 3.0.2__tar.gz
This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
- {arrowbricks-3.0.1 → arrowbricks-3.0.2}/PKG-INFO +1 -1
- {arrowbricks-3.0.1 → arrowbricks-3.0.2}/pyproject.toml +1 -1
- {arrowbricks-3.0.1 → arrowbricks-3.0.2}/rust/arrowbricks_core/Cargo.lock +1 -1
- {arrowbricks-3.0.1 → arrowbricks-3.0.2}/rust/arrowbricks_core/Cargo.toml +15 -3
- {arrowbricks-3.0.1 → arrowbricks-3.0.2}/rust/arrowbricks_core/src/client.rs +176 -185
- {arrowbricks-3.0.1 → arrowbricks-3.0.2}/rust/arrowbricks_core/src/heartbeat.rs +20 -42
- {arrowbricks-3.0.1 → arrowbricks-3.0.2}/rust/arrowbricks_core/src/json_convert.rs +8 -25
- {arrowbricks-3.0.1 → arrowbricks-3.0.2}/rust/arrowbricks_core/src/lib.rs +64 -41
- {arrowbricks-3.0.1 → arrowbricks-3.0.2}/rust/arrowbricks_core/src/pipeline.rs +204 -194
- {arrowbricks-3.0.1 → arrowbricks-3.0.2}/rust/arrowbricks_core/src/thrift.rs +19 -28
- {arrowbricks-3.0.1 → arrowbricks-3.0.2}/rust/arrowbricks_core/tests/wiremock_pipeline.rs +1 -164
- {arrowbricks-3.0.1 → arrowbricks-3.0.2}/rust/arrowbricks_core/tests/wiremock_thrift.rs +126 -2
- {arrowbricks-3.0.1 → arrowbricks-3.0.2}/rust/arrowbricks_core/tests_py/test_token_provider.py +17 -0
- {arrowbricks-3.0.1 → arrowbricks-3.0.2}/src/arrowbricks/_streaming.py +2 -3
- {arrowbricks-3.0.1 → arrowbricks-3.0.2}/src/arrowbricks/client.py +9 -25
- {arrowbricks-3.0.1 → arrowbricks-3.0.2}/src/arrowbricks/cursor.py +70 -20
- {arrowbricks-3.0.1 → arrowbricks-3.0.2}/LICENSE +0 -0
- {arrowbricks-3.0.1 → arrowbricks-3.0.2}/README.md +0 -0
- {arrowbricks-3.0.1 → arrowbricks-3.0.2}/rust/arrowbricks_core/.gitignore +0 -0
- {arrowbricks-3.0.1 → arrowbricks-3.0.2}/rust/arrowbricks_core/README.md +0 -0
- {arrowbricks-3.0.1 → arrowbricks-3.0.2}/rust/arrowbricks_core/examples/duckdb_query.py +0 -0
- {arrowbricks-3.0.1 → arrowbricks-3.0.2}/rust/arrowbricks_core/examples/fastapi_sse.py +0 -0
- {arrowbricks-3.0.1 → arrowbricks-3.0.2}/rust/arrowbricks_core/rustfmt.toml +0 -0
- {arrowbricks-3.0.1 → arrowbricks-3.0.2}/rust/arrowbricks_core/tests/wiremock_volume_files.rs +0 -0
- {arrowbricks-3.0.1 → arrowbricks-3.0.2}/rust/arrowbricks_core/tests_py/conftest.py +0 -0
- {arrowbricks-3.0.1 → arrowbricks-3.0.2}/rust/arrowbricks_core/tests_py/test_ipc_stream.py +0 -0
- {arrowbricks-3.0.1 → arrowbricks-3.0.2}/rust/arrowbricks_core/tests_py/test_parameters.py +0 -0
- {arrowbricks-3.0.1 → arrowbricks-3.0.2}/rust/arrowbricks_core/tests_py/test_stream_ndjson_lines.py +0 -0
- {arrowbricks-3.0.1 → arrowbricks-3.0.2}/rust/arrowbricks_core/tests_py/test_streaming.py +0 -0
- {arrowbricks-3.0.1 → arrowbricks-3.0.2}/rust/arrowbricks_core/tests_py/test_thrift.py +0 -0
- {arrowbricks-3.0.1 → arrowbricks-3.0.2}/rust/arrowbricks_core/tests_py/test_volume_files.py +0 -0
- {arrowbricks-3.0.1 → arrowbricks-3.0.2}/rust/arrowbricks_core/tests_py/thrift_mock.py +0 -0
- {arrowbricks-3.0.1 → arrowbricks-3.0.2}/src/arrowbricks/__init__.py +0 -0
- {arrowbricks-3.0.1 → arrowbricks-3.0.2}/src/arrowbricks/_core.pyi +0 -0
- {arrowbricks-3.0.1 → arrowbricks-3.0.2}/src/arrowbricks/py.typed +0 -0
|
@@ -1,6 +1,6 @@
|
|
|
1
1
|
[project]
|
|
2
2
|
name = "arrowbricks"
|
|
3
|
-
version = "3.0.
|
|
3
|
+
version = "3.0.2"
|
|
4
4
|
description = "Runs SQL against a Databricks SQL warehouse via the Statement Execution API and hands you the result as Arrow -- a DB-API-ish Cursor (fetchone/fetchmany/fetchall/fetchall_arrow) or NDJSON streaming. Rust/PyO3 core throughout -- zero required runtime dependencies."
|
|
5
5
|
readme = "README.md"
|
|
6
6
|
license = "MIT"
|
|
@@ -1,6 +1,6 @@
|
|
|
1
1
|
[package]
|
|
2
2
|
name = "arrowbricks_core"
|
|
3
|
-
version = "3.0.
|
|
3
|
+
version = "3.0.2"
|
|
4
4
|
edition = "2024"
|
|
5
5
|
readme = "README.md"
|
|
6
6
|
|
|
@@ -23,8 +23,13 @@ chrono = { version = "0.4", default-features = false, features = ["std"] }
|
|
|
23
23
|
# Decompresses cloud-fetch chunk bytes when the server honors our
|
|
24
24
|
# `result_compression: LZ4_FRAME` request (see client.rs's execute_statement)
|
|
25
25
|
# -- the same LZ4 Frame format databricks-sql-python's `lz4.frame` decodes.
|
|
26
|
-
# Pure Rust
|
|
27
|
-
#
|
|
26
|
+
# Pure Rust, no C toolchain dependency -- unlike reqwest's `rustls` feature
|
|
27
|
+
# below, which *does* pull one in (`__rustls-aws-lc-rs` -> aws-lc-rs ->
|
|
28
|
+
# aws-lc-sys, confirmed via `cargo tree -i aws-lc-sys`; this crate has no
|
|
29
|
+
# `ring` feature available in reqwest 0.13 to avoid it with). A previous
|
|
30
|
+
# version of this comment claimed the opposite -- that this crate avoids
|
|
31
|
+
# aws-lc-sys-style native-build pain elsewhere too -- found wrong on review;
|
|
32
|
+
# see AGENTS.md's build-prerequisites note.
|
|
28
33
|
lz4_flex = { version = "0.11", default-features = false, features = ["frame", "std"] }
|
|
29
34
|
pyo3 = { version = "0.29.1", features = ["abi3-py311"] }
|
|
30
35
|
# pyo3-arrow's only optional feature is buffer_protocol (on by default),
|
|
@@ -53,6 +58,13 @@ extension-module = ["pyo3/extension-module"]
|
|
|
53
58
|
tokio = { version = "1.53.1", features = ["rt-multi-thread", "macros", "test-util", "net", "io-util"] }
|
|
54
59
|
wiremock = "0.6.5"
|
|
55
60
|
|
|
61
|
+
# Currently clean on all three by luck, not enforcement -- pinning them here
|
|
62
|
+
# so a future PR that introduces one fails CI instead of silently landing.
|
|
63
|
+
[lints.clippy]
|
|
64
|
+
redundant_clone = "warn"
|
|
65
|
+
large_enum_variant = "warn"
|
|
66
|
+
needless_collect = "warn"
|
|
67
|
+
|
|
56
68
|
[profile.release]
|
|
57
69
|
# opt-level stays at the implicit default (3, speed) rather than "s"/"z" --
|
|
58
70
|
# perf was the original ask and the benchmarks were run against opt-level 3;
|
|
@@ -148,8 +148,7 @@ struct ResultBody {
|
|
|
148
148
|
external_links: Vec<ResultLinkBody>,
|
|
149
149
|
/// Only present for `disposition: INLINE` + `format: JSON_ARRAY` -- each
|
|
150
150
|
/// row a `Vec<Option<String>>` (every non-null value a string,
|
|
151
|
-
/// Databricks' own JSON_ARRAY contract
|
|
152
|
-
/// normal `EXTERNAL_LINKS`+`JSON_ARRAY` chunks). Consumed by
|
|
151
|
+
/// Databricks' own JSON_ARRAY contract). Consumed by
|
|
153
152
|
/// `json_convert::json_array_to_record_batch` on the `prefer_inline`
|
|
154
153
|
/// fast path.
|
|
155
154
|
#[serde(default)]
|
|
@@ -329,7 +328,7 @@ pub struct StatementSubmitResult {
|
|
|
329
328
|
|
|
330
329
|
/// What `execute_arrow_statement_prefer_inline` gets you -- either the whole
|
|
331
330
|
/// result inline as raw JSON_ARRAY rows (every non-null value a string,
|
|
332
|
-
///
|
|
331
|
+
/// Databricks' own JSON_ARRAY contract; converting these into a real
|
|
333
332
|
/// `RecordBatch` is `pipeline.rs`'s job via `json_convert`, not this
|
|
334
333
|
/// Arrow-agnostic module's -- see this file's own module doc comment), or a
|
|
335
334
|
/// normal `StatementSubmitResult` to fetch chunks for exactly as if
|
|
@@ -353,15 +352,18 @@ pub struct ChunkItem {
|
|
|
353
352
|
pub blob: Bytes,
|
|
354
353
|
pub row_count: Option<i64>,
|
|
355
354
|
pub chunk_index: i64,
|
|
356
|
-
///
|
|
357
|
-
///
|
|
358
|
-
/// decoded batches must be sliced down to if they exceed it, since a
|
|
359
|
-
/// Thrift `resultLinks` file (like SEA's own cloud-fetch files) can
|
|
355
|
+
/// A declared row-count bound this chunk's own decoded batches must be
|
|
356
|
+
/// sliced down to if they exceed it, since a cloud-fetch file can
|
|
360
357
|
/// legitimately contain more rows than its own declared count for a
|
|
361
358
|
/// `LIMIT`-bounded query (see `pipeline.rs`'s `decode_chunk_item`).
|
|
362
|
-
///
|
|
363
|
-
///
|
|
364
|
-
///
|
|
359
|
+
/// Every real producer sets this now -- Thrift `resultLinks`
|
|
360
|
+
/// (`pipeline.rs`'s `fetch_thrift_link`), Thrift's inline `arrowBatches`
|
|
361
|
+
/// (`pipeline.rs`'s `run_thrift_fetch_loop`), and SEA's own chunk fetch
|
|
362
|
+
/// (`fetch_chunks_with_backpressure`, below, though only when a
|
|
363
|
+
/// `chunk_index` resolves to exactly one blob -- see that function's own
|
|
364
|
+
/// comment for why more than one blob is left untruncated). `None`
|
|
365
|
+
/// means no authoritative bound is known, not "this chunk was never
|
|
366
|
+
/// overshot" -- `decode_chunk_item` treats it as a no-op either way.
|
|
365
367
|
pub truncate_to: Option<i64>,
|
|
366
368
|
}
|
|
367
369
|
|
|
@@ -507,9 +509,21 @@ pub struct DbClient {
|
|
|
507
509
|
/// rebuild the extension to do it. Also doubles, on the Thrift path, as
|
|
508
510
|
/// `TExecuteStatementReq.canDecompressLZ4Result`.
|
|
509
511
|
compress_results: bool,
|
|
510
|
-
session_pool:
|
|
512
|
+
session_pool: Pool<String>,
|
|
511
513
|
pub protocol: Protocol,
|
|
512
|
-
|
|
514
|
+
/// Same checkout/checkin shape as `session_pool` above, plus: Thrift's
|
|
515
|
+
/// session model is *mandatory* per statement, unlike SEA's optional
|
|
516
|
+
/// `session_id`, and this crate has not independently confirmed whether
|
|
517
|
+
/// concurrent statements on one Thrift session are safe against a real
|
|
518
|
+
/// workspace the way the SEA crash was -- so this pool exists
|
|
519
|
+
/// defensively, following the same proven-safe pattern regardless of
|
|
520
|
+
/// whether the analogous crash reproduces here. Unlike SEA, there is no
|
|
521
|
+
/// "sessionless" fallback available at all -- `TExecuteStatementReq.
|
|
522
|
+
/// sessionHandle` is required by the protocol -- so a pool-exhaustion/
|
|
523
|
+
/// creation-failure `None` from `thrift_checkout_session` means the
|
|
524
|
+
/// caller must open one throwaway session for that single call and
|
|
525
|
+
/// close it again immediately after (see `execute_lazy_thrift`).
|
|
526
|
+
thrift_session_pool: Pool<thrift::SessionHandle>,
|
|
513
527
|
/// Global budget for concurrent cloud-fetch HTTP requests, sized to
|
|
514
528
|
/// `chunk_fetch_concurrency`. Every link download takes one permit; a
|
|
515
529
|
/// download that finds spare permits (few links in flight -- exactly
|
|
@@ -529,24 +543,12 @@ pub struct DbClient {
|
|
|
529
543
|
/// ever split into -- see `DbClient::download_slots`.
|
|
530
544
|
pub const MAX_SPLIT_PARTS: usize = 8;
|
|
531
545
|
|
|
532
|
-
///
|
|
533
|
-
///
|
|
534
|
-
///
|
|
535
|
-
///
|
|
536
|
-
///
|
|
537
|
-
|
|
538
|
-
/// workspace the way the SEA crash was -- so this pool exists defensively,
|
|
539
|
-
/// following the same proven-safe pattern regardless of whether the
|
|
540
|
-
/// analogous crash reproduces here. Unlike SEA, there is no "sessionless"
|
|
541
|
-
/// fallback available at all -- `TExecuteStatementReq.sessionHandle` is
|
|
542
|
-
/// required by the protocol -- so a pool-exhaustion/creation-failure `None`
|
|
543
|
-
/// here means the caller must open one throwaway session for that single
|
|
544
|
-
/// call and close it again immediately after (see `execute_lazy_thrift`).
|
|
545
|
-
#[derive(Default)]
|
|
546
|
-
struct ThriftSessionPool {
|
|
547
|
-
idle: Mutex<HashMap<(Option<String>, Option<String>), Vec<thrift::SessionHandle>>>,
|
|
548
|
-
total: Mutex<HashMap<(Option<String>, Option<String>), usize>>,
|
|
549
|
-
}
|
|
546
|
+
/// `chunk_fetch_concurrency`'s own default -- see `DbClient::new`'s own
|
|
547
|
+
/// comment on it for the measured numbers behind picking 64. Named so
|
|
548
|
+
/// `download_slots` (sized to this same number, since it's a budget over
|
|
549
|
+
/// the same worker count) can't drift from it the way a second bare `64`
|
|
550
|
+
/// silently could.
|
|
551
|
+
const DEFAULT_CHUNK_FETCH_CONCURRENCY: usize = 64;
|
|
550
552
|
|
|
551
553
|
/// How long the Thrift path polls `GetOperationStatus` when a statement
|
|
552
554
|
/// doesn't finish within its `getDirectResults` budget (see
|
|
@@ -621,13 +623,77 @@ const THRIFT_DIRECT_RESULTS_MAX_BYTES: i64 = 1024 * 1024 * 1024;
|
|
|
621
623
|
/// still throws away a perfectly good session), but guarantees a session
|
|
622
624
|
/// that might be in the same bad state behind the SparkSession-null crash
|
|
623
625
|
/// above is never handed to a second caller.
|
|
624
|
-
|
|
625
|
-
|
|
626
|
+
type SessionKey = (Option<String>, Option<String>);
|
|
627
|
+
|
|
628
|
+
/// Generic checkout/checkin pool keyed by (catalog, schema) -- shared by
|
|
629
|
+
/// SEA's session ids (`Pool<String>`) and Thrift's session handles
|
|
630
|
+
/// (`Pool<thrift::SessionHandle>`), which had this exact logic duplicated
|
|
631
|
+
/// twice over before this. See `checkout_session`/`thrift_checkout_session`
|
|
632
|
+
/// for the constraints that shape it (session pinned to one (catalog,
|
|
633
|
+
/// schema) pair, safe for sequential reuse only, checkout never blocks on
|
|
634
|
+
/// exhaustion).
|
|
635
|
+
struct Pool<T> {
|
|
626
636
|
// ponytail: fixed cap, not a constructor kwarg -- nothing's asked to
|
|
627
637
|
// tune this yet; raise (or expose one) if a workload needs more
|
|
628
638
|
// concurrent sessions per (catalog, schema) pair than this.
|
|
629
|
-
idle: Mutex<HashMap<
|
|
630
|
-
total: Mutex<HashMap<
|
|
639
|
+
idle: Mutex<HashMap<SessionKey, Vec<T>>>,
|
|
640
|
+
total: Mutex<HashMap<SessionKey, usize>>,
|
|
641
|
+
}
|
|
642
|
+
|
|
643
|
+
// Manual impl instead of `#[derive(Default)]`: the derive would wrongly
|
|
644
|
+
// require `T: Default` even though neither field actually needs it.
|
|
645
|
+
impl<T> Default for Pool<T> {
|
|
646
|
+
fn default() -> Self {
|
|
647
|
+
Self {
|
|
648
|
+
idle: Mutex::new(HashMap::new()),
|
|
649
|
+
total: Mutex::new(HashMap::new()),
|
|
650
|
+
}
|
|
651
|
+
}
|
|
652
|
+
}
|
|
653
|
+
|
|
654
|
+
impl<T> Pool<T> {
|
|
655
|
+
fn take(&self, key: &SessionKey) -> Option<T> {
|
|
656
|
+
self.idle.lock().unwrap().get_mut(key).and_then(Vec::pop)
|
|
657
|
+
}
|
|
658
|
+
|
|
659
|
+
/// Reserves a slot for `key` if under `MAX_SESSIONS_PER_KEY`. The
|
|
660
|
+
/// increment happens before the caller's own creation call so two
|
|
661
|
+
/// concurrent callers can't both squeeze past the cap; call `release` if
|
|
662
|
+
/// creation then fails.
|
|
663
|
+
fn reserve(&self, key: &SessionKey) -> bool {
|
|
664
|
+
let mut total = self.total.lock().unwrap();
|
|
665
|
+
let count = total.entry(key.clone()).or_insert(0);
|
|
666
|
+
if *count >= MAX_SESSIONS_PER_KEY {
|
|
667
|
+
return false;
|
|
668
|
+
}
|
|
669
|
+
*count += 1;
|
|
670
|
+
true
|
|
671
|
+
}
|
|
672
|
+
|
|
673
|
+
fn release(&self, key: &SessionKey) {
|
|
674
|
+
let mut total = self.total.lock().unwrap();
|
|
675
|
+
if let Some(count) = total.get_mut(key) {
|
|
676
|
+
*count = count.saturating_sub(1);
|
|
677
|
+
}
|
|
678
|
+
}
|
|
679
|
+
|
|
680
|
+
fn put(&self, key: SessionKey, item: T) {
|
|
681
|
+
self.idle.lock().unwrap().entry(key).or_default().push(item);
|
|
682
|
+
}
|
|
683
|
+
|
|
684
|
+
/// Returns `item` to the pool for reuse (`keep = true`) or discards it,
|
|
685
|
+
/// releasing its reservation instead (`keep = false`).
|
|
686
|
+
fn checkin(&self, key: SessionKey, item: T, keep: bool) {
|
|
687
|
+
if keep {
|
|
688
|
+
self.put(key, item);
|
|
689
|
+
} else {
|
|
690
|
+
self.release(&key);
|
|
691
|
+
}
|
|
692
|
+
}
|
|
693
|
+
|
|
694
|
+
fn drain_idle(&self) -> Vec<T> {
|
|
695
|
+
self.idle.lock().unwrap().drain().flat_map(|(_, v)| v).collect()
|
|
696
|
+
}
|
|
631
697
|
}
|
|
632
698
|
|
|
633
699
|
pub const MAX_SESSIONS_PER_KEY: usize = 8;
|
|
@@ -672,15 +738,15 @@ impl DbClient {
|
|
|
672
738
|
// a claim that 64 is the true optimum. A caller with a very
|
|
673
739
|
// large chunk count or a fast/low-latency link to the warehouse
|
|
674
740
|
// may still want to raise it further.
|
|
675
|
-
chunk_fetch_concurrency:
|
|
741
|
+
chunk_fetch_concurrency: DEFAULT_CHUNK_FETCH_CONCURRENCY,
|
|
676
742
|
warehouse_start_timeout: Duration::from_secs(300),
|
|
677
743
|
warehouse_confirmed_running_ttl: Duration::from_secs(30),
|
|
678
744
|
warehouse_confirmed_running_at: Mutex::new(None),
|
|
679
745
|
compress_results: true,
|
|
680
|
-
session_pool:
|
|
746
|
+
session_pool: Pool::default(),
|
|
681
747
|
protocol: Protocol::Thrift,
|
|
682
|
-
thrift_session_pool:
|
|
683
|
-
download_slots: tokio::sync::Semaphore::new(
|
|
748
|
+
thrift_session_pool: Pool::default(),
|
|
749
|
+
download_slots: tokio::sync::Semaphore::new(DEFAULT_CHUNK_FETCH_CONCURRENCY),
|
|
684
750
|
}
|
|
685
751
|
}
|
|
686
752
|
|
|
@@ -690,11 +756,7 @@ impl DbClient {
|
|
|
690
756
|
self
|
|
691
757
|
}
|
|
692
758
|
|
|
693
|
-
/// Selects the wire protocol/backend -- `Protocol
|
|
694
|
-
/// `DbClient` constructor's own internal starting value, see
|
|
695
|
-
/// `Protocol`'s own doc comment for why that's not the same thing as
|
|
696
|
-
/// "the user-facing default") or `Protocol::Thrift` (the actual
|
|
697
|
-
/// user-facing default as of this session's benchmarking work).
|
|
759
|
+
/// Selects the wire protocol/backend -- see `Protocol`'s own doc comment.
|
|
698
760
|
pub fn with_protocol(mut self, protocol: Protocol) -> Self {
|
|
699
761
|
self.protocol = protocol;
|
|
700
762
|
self
|
|
@@ -830,11 +892,9 @@ impl DbClient {
|
|
|
830
892
|
}
|
|
831
893
|
|
|
832
894
|
/// Downloads one cloud-fetch link, splitting it across parallel HTTP
|
|
833
|
-
/// Range requests when
|
|
834
|
-
///
|
|
835
|
-
///
|
|
836
|
-
/// stream leaves most of the link idle. See `download_slots`' own doc
|
|
837
|
-
/// comment for the measured win and why the large-result case is safe.
|
|
895
|
+
/// Range requests when the shared `download_slots` budget has room to
|
|
896
|
+
/// spare -- see that field's own doc comment for the measured win and
|
|
897
|
+
/// why the large-result case is safe.
|
|
838
898
|
pub(crate) async fn fetch_link_bytes_budgeted(
|
|
839
899
|
self: &Arc<Self>,
|
|
840
900
|
url: &str,
|
|
@@ -843,8 +903,16 @@ impl DbClient {
|
|
|
843
903
|
// One permit per download is mandatory; the worker pool already
|
|
844
904
|
// bounds concurrent links to `chunk_fetch_concurrency`, so this
|
|
845
905
|
// never blocks in practice -- it just makes the budget accounting
|
|
846
|
-
// exact.
|
|
847
|
-
|
|
906
|
+
// exact. `acquire()`'s `Err` case (the semaphore closed) can't
|
|
907
|
+
// happen -- nothing in this crate ever calls `.close()` on
|
|
908
|
+
// `download_slots` -- but propagating it as a real error instead of
|
|
909
|
+
// an `.expect()` costs nothing and avoids a panic if that ever
|
|
910
|
+
// changes.
|
|
911
|
+
let _base = self
|
|
912
|
+
.download_slots
|
|
913
|
+
.acquire()
|
|
914
|
+
.await
|
|
915
|
+
.map_err(|e| ApiError::permanent(format!("download_slots semaphore closed unexpectedly: {e}")))?;
|
|
848
916
|
let want = (MAX_SPLIT_PARTS - 1).min(self.download_slots.available_permits());
|
|
849
917
|
let extra = if want > 0 {
|
|
850
918
|
self.download_slots.try_acquire_many(want as u32).ok()
|
|
@@ -960,8 +1028,33 @@ impl DbClient {
|
|
|
960
1028
|
|
|
961
1029
|
let mut out = bytes::BytesMut::with_capacity(total.unwrap_or(head.len() as u64) as usize);
|
|
962
1030
|
out.extend_from_slice(&head);
|
|
963
|
-
|
|
964
|
-
|
|
1031
|
+
// Parts must concatenate in order, so this can't use `JoinSet`
|
|
1032
|
+
// (which yields in completion order) without tracking indices --
|
|
1033
|
+
// simpler to keep the `Vec` and explicitly `.abort()` every
|
|
1034
|
+
// not-yet-awaited sibling the moment one part fails, rather than
|
|
1035
|
+
// silently leaving them running (a bare `?` here would return
|
|
1036
|
+
// early and just drop the rest, which does NOT cancel them --
|
|
1037
|
+
// `JoinHandle::drop` detaches, it doesn't abort -- leaving up to
|
|
1038
|
+
// `MAX_SPLIT_PARTS - 1` sibling Range downloads, each with their
|
|
1039
|
+
// own `retry_call` backoff, still in flight for a link the caller
|
|
1040
|
+
// has already given up on).
|
|
1041
|
+
let mut iter = handles.into_iter();
|
|
1042
|
+
while let Some(h) = iter.next() {
|
|
1043
|
+
match h.await {
|
|
1044
|
+
Ok(Ok(part)) => out.extend_from_slice(&part),
|
|
1045
|
+
Ok(Err(e)) => {
|
|
1046
|
+
for remaining in iter {
|
|
1047
|
+
remaining.abort();
|
|
1048
|
+
}
|
|
1049
|
+
return Err(e);
|
|
1050
|
+
}
|
|
1051
|
+
Err(join_err) => {
|
|
1052
|
+
for remaining in iter {
|
|
1053
|
+
remaining.abort();
|
|
1054
|
+
}
|
|
1055
|
+
return Err(join_error(join_err));
|
|
1056
|
+
}
|
|
1057
|
+
}
|
|
965
1058
|
}
|
|
966
1059
|
let bytes = out.freeze();
|
|
967
1060
|
if let Some(t) = total
|
|
@@ -981,19 +1074,15 @@ impl DbClient {
|
|
|
981
1074
|
}
|
|
982
1075
|
|
|
983
1076
|
/// Shared by both protocols -- plain REST against `/api/2.0/sql/warehouses/{id}`,
|
|
984
|
-
/// nothing SEA- or Thrift-specific about it.
|
|
985
|
-
///
|
|
986
|
-
/// didn't, which meant a stopped warehouse got no proactive wake on
|
|
987
|
-
/// `protocol="thrift"` (now the default) -- statement submission would
|
|
988
|
-
/// eventually surface an error instead, with no `warehouse_start_timeout`
|
|
989
|
-
/// wait for it to come up first. Fixed by calling this from both.
|
|
1077
|
+
/// nothing SEA- or Thrift-specific about it (see AGENTS.md for why the
|
|
1078
|
+
/// Thrift path didn't always call this).
|
|
990
1079
|
pub(crate) async fn ensure_warehouse_running(&self) -> Result<(), ApiError> {
|
|
991
1080
|
{
|
|
992
1081
|
let confirmed = *self.warehouse_confirmed_running_at.lock().unwrap();
|
|
993
|
-
if let Some(at) = confirmed
|
|
994
|
-
|
|
995
|
-
|
|
996
|
-
|
|
1082
|
+
if let Some(at) = confirmed
|
|
1083
|
+
&& at.elapsed() < self.warehouse_confirmed_running_ttl
|
|
1084
|
+
{
|
|
1085
|
+
return Ok(());
|
|
997
1086
|
}
|
|
998
1087
|
}
|
|
999
1088
|
|
|
@@ -1048,36 +1137,20 @@ impl DbClient {
|
|
|
1048
1137
|
/// Hands back a pooled session for (`catalog`, `schema`) if one's idle,
|
|
1049
1138
|
/// creates one if the pool for that key isn't at `MAX_SESSIONS_PER_KEY`
|
|
1050
1139
|
/// yet, or `None` if neither -- the caller falls back to a plain
|
|
1051
|
-
/// session-less submission in that case, see `
|
|
1052
|
-
/// comment for why this never blocks instead.
|
|
1053
|
-
/// increment happens *before* the `create_session` await so two
|
|
1054
|
-
/// concurrent callers can't both squeeze past the cap; a failed creation
|
|
1055
|
-
/// releases the reservation again.
|
|
1140
|
+
/// session-less submission in that case, see `session_pool`'s own doc
|
|
1141
|
+
/// comment for why this never blocks instead.
|
|
1056
1142
|
async fn checkout_session(&self, catalog: Option<&str>, schema: Option<&str>) -> Option<String> {
|
|
1057
1143
|
let key = (catalog.map(str::to_string), schema.map(str::to_string));
|
|
1058
|
-
{
|
|
1059
|
-
|
|
1060
|
-
if let Some(ids) = idle.get_mut(&key)
|
|
1061
|
-
&& let Some(id) = ids.pop()
|
|
1062
|
-
{
|
|
1063
|
-
return Some(id);
|
|
1064
|
-
}
|
|
1144
|
+
if let Some(id) = self.session_pool.take(&key) {
|
|
1145
|
+
return Some(id);
|
|
1065
1146
|
}
|
|
1066
|
-
{
|
|
1067
|
-
|
|
1068
|
-
let count = total.entry(key.clone()).or_insert(0);
|
|
1069
|
-
if *count >= MAX_SESSIONS_PER_KEY {
|
|
1070
|
-
return None;
|
|
1071
|
-
}
|
|
1072
|
-
*count += 1;
|
|
1147
|
+
if !self.session_pool.reserve(&key) {
|
|
1148
|
+
return None;
|
|
1073
1149
|
}
|
|
1074
1150
|
match self.create_session(catalog, schema).await {
|
|
1075
1151
|
Ok(id) => Some(id),
|
|
1076
1152
|
Err(_) => {
|
|
1077
|
-
|
|
1078
|
-
if let Some(count) = total.get_mut(&key) {
|
|
1079
|
-
*count = count.saturating_sub(1);
|
|
1080
|
-
}
|
|
1153
|
+
self.session_pool.release(&key);
|
|
1081
1154
|
None
|
|
1082
1155
|
}
|
|
1083
1156
|
}
|
|
@@ -1085,24 +1158,11 @@ impl DbClient {
|
|
|
1085
1158
|
|
|
1086
1159
|
/// Returns a session to the pool for reuse (`keep = true`, the statement
|
|
1087
1160
|
/// it backed reached SUCCEEDED/FAILED/CANCELED cleanly) or discards it
|
|
1088
|
-
/// (`keep = false`) -- see `
|
|
1161
|
+
/// (`keep = false`) -- see `session_pool`'s doc comment for why any error
|
|
1089
1162
|
/// discards rather than reuses.
|
|
1090
1163
|
fn checkin_session(&self, catalog: Option<&str>, schema: Option<&str>, session_id: String, keep: bool) {
|
|
1091
1164
|
let key = (catalog.map(str::to_string), schema.map(str::to_string));
|
|
1092
|
-
|
|
1093
|
-
self.session_pool
|
|
1094
|
-
.idle
|
|
1095
|
-
.lock()
|
|
1096
|
-
.unwrap()
|
|
1097
|
-
.entry(key)
|
|
1098
|
-
.or_default()
|
|
1099
|
-
.push(session_id);
|
|
1100
|
-
} else {
|
|
1101
|
-
let mut total = self.session_pool.total.lock().unwrap();
|
|
1102
|
-
if let Some(count) = total.get_mut(&key) {
|
|
1103
|
-
*count = count.saturating_sub(1);
|
|
1104
|
-
}
|
|
1105
|
-
}
|
|
1165
|
+
self.session_pool.checkin(key, session_id, keep);
|
|
1106
1166
|
}
|
|
1107
1167
|
|
|
1108
1168
|
/// Best-effort cleanup of every currently-idle pooled session -- meant
|
|
@@ -1113,11 +1173,7 @@ impl DbClient {
|
|
|
1113
1173
|
/// this before every pending statement has finished is a caller
|
|
1114
1174
|
/// ordering issue, not something this method can fix from inside.
|
|
1115
1175
|
pub async fn close_all_sessions(&self) {
|
|
1116
|
-
|
|
1117
|
-
let mut idle = self.session_pool.idle.lock().unwrap();
|
|
1118
|
-
idle.drain().flat_map(|(_, v)| v).collect()
|
|
1119
|
-
};
|
|
1120
|
-
for id in ids {
|
|
1176
|
+
for id in self.session_pool.drain_idle() {
|
|
1121
1177
|
self.delete_session(&id).await;
|
|
1122
1178
|
}
|
|
1123
1179
|
}
|
|
@@ -1208,8 +1264,8 @@ impl DbClient {
|
|
|
1208
1264
|
}
|
|
1209
1265
|
|
|
1210
1266
|
/// Same checkout contract as SEA's `checkout_session` (see
|
|
1211
|
-
/// `
|
|
1212
|
-
/// its own throwaway session for this one call (Thrift has no
|
|
1267
|
+
/// `thrift_session_pool`'s doc comment): `None` means the caller must
|
|
1268
|
+
/// open its own throwaway session for this one call (Thrift has no
|
|
1213
1269
|
/// session-less submission mode to fall back to).
|
|
1214
1270
|
pub(crate) async fn thrift_checkout_session(
|
|
1215
1271
|
&self,
|
|
@@ -1217,29 +1273,16 @@ impl DbClient {
|
|
|
1217
1273
|
schema: Option<&str>,
|
|
1218
1274
|
) -> Option<thrift::SessionHandle> {
|
|
1219
1275
|
let key = (catalog.map(str::to_string), schema.map(str::to_string));
|
|
1220
|
-
{
|
|
1221
|
-
|
|
1222
|
-
if let Some(ids) = idle.get_mut(&key)
|
|
1223
|
-
&& let Some(id) = ids.pop()
|
|
1224
|
-
{
|
|
1225
|
-
return Some(id);
|
|
1226
|
-
}
|
|
1276
|
+
if let Some(handle) = self.thrift_session_pool.take(&key) {
|
|
1277
|
+
return Some(handle);
|
|
1227
1278
|
}
|
|
1228
|
-
{
|
|
1229
|
-
|
|
1230
|
-
let count = total.entry(key.clone()).or_insert(0);
|
|
1231
|
-
if *count >= MAX_SESSIONS_PER_KEY {
|
|
1232
|
-
return None;
|
|
1233
|
-
}
|
|
1234
|
-
*count += 1;
|
|
1279
|
+
if !self.thrift_session_pool.reserve(&key) {
|
|
1280
|
+
return None;
|
|
1235
1281
|
}
|
|
1236
1282
|
match self.thrift_open_session_raw(catalog, schema).await {
|
|
1237
1283
|
Ok(handle) => Some(handle),
|
|
1238
1284
|
Err(_) => {
|
|
1239
|
-
|
|
1240
|
-
if let Some(count) = total.get_mut(&key) {
|
|
1241
|
-
*count = count.saturating_sub(1);
|
|
1242
|
-
}
|
|
1285
|
+
self.thrift_session_pool.release(&key);
|
|
1243
1286
|
None
|
|
1244
1287
|
}
|
|
1245
1288
|
}
|
|
@@ -1253,30 +1296,13 @@ impl DbClient {
|
|
|
1253
1296
|
keep: bool,
|
|
1254
1297
|
) {
|
|
1255
1298
|
let key = (catalog.map(str::to_string), schema.map(str::to_string));
|
|
1256
|
-
|
|
1257
|
-
self.thrift_session_pool
|
|
1258
|
-
.idle
|
|
1259
|
-
.lock()
|
|
1260
|
-
.unwrap()
|
|
1261
|
-
.entry(key)
|
|
1262
|
-
.or_default()
|
|
1263
|
-
.push(session);
|
|
1264
|
-
} else {
|
|
1265
|
-
let mut total = self.thrift_session_pool.total.lock().unwrap();
|
|
1266
|
-
if let Some(count) = total.get_mut(&key) {
|
|
1267
|
-
*count = count.saturating_sub(1);
|
|
1268
|
-
}
|
|
1269
|
-
}
|
|
1299
|
+
self.thrift_session_pool.checkin(key, session, keep);
|
|
1270
1300
|
}
|
|
1271
1301
|
|
|
1272
1302
|
/// Best-effort close of every currently-idle pooled Thrift session --
|
|
1273
1303
|
/// same contract as `close_all_sessions` (SEA).
|
|
1274
1304
|
pub async fn close_all_thrift_sessions(&self) {
|
|
1275
|
-
|
|
1276
|
-
let mut idle = self.thrift_session_pool.idle.lock().unwrap();
|
|
1277
|
-
idle.drain().flat_map(|(_, v)| v).collect()
|
|
1278
|
-
};
|
|
1279
|
-
for s in sessions {
|
|
1305
|
+
for s in self.thrift_session_pool.drain_idle() {
|
|
1280
1306
|
self.thrift_close_session_raw(&s).await;
|
|
1281
1307
|
}
|
|
1282
1308
|
}
|
|
@@ -1312,7 +1338,6 @@ impl DbClient {
|
|
|
1312
1338
|
op: &thrift::OperationHandle,
|
|
1313
1339
|
) -> Result<thrift::OperationStatusResp, ApiError> {
|
|
1314
1340
|
let body = Bytes::from(thrift::build_get_operation_status(op));
|
|
1315
|
-
// Idempotent -- read-only.
|
|
1316
1341
|
let resp_bytes = self.thrift_call(body, true).await?;
|
|
1317
1342
|
thrift::parse_get_operation_status(&resp_bytes).map_err(Self::thrift_parse_error)
|
|
1318
1343
|
}
|
|
@@ -1445,24 +1470,6 @@ impl DbClient {
|
|
|
1445
1470
|
}
|
|
1446
1471
|
}
|
|
1447
1472
|
|
|
1448
|
-
/// Like `execute_arrow_statement`, fixed to JSON_ARRAY -- each fetched
|
|
1449
|
-
/// chunk's bytes are then a JSON array of rows, each row itself an array
|
|
1450
|
-
/// of values where every non-null value is a *string* regardless of its
|
|
1451
|
-
/// real column type (Databricks' own JSON_ARRAY contract; null stays
|
|
1452
|
-
/// JSON null) -- casting by the manifest's column type_name, if wanted,
|
|
1453
|
-
/// is left to the caller, same as the Python original does nothing extra
|
|
1454
|
-
/// here either.
|
|
1455
|
-
pub async fn execute_json_statement(
|
|
1456
|
-
&self,
|
|
1457
|
-
statement: &str,
|
|
1458
|
-
catalog: Option<&str>,
|
|
1459
|
-
schema: Option<&str>,
|
|
1460
|
-
parameters: Option<Value>,
|
|
1461
|
-
) -> Result<StatementSubmitResult, ApiError> {
|
|
1462
|
-
self.execute_statement(statement, "JSON_ARRAY", catalog, schema, parameters)
|
|
1463
|
-
.await
|
|
1464
|
-
}
|
|
1465
|
-
|
|
1466
1473
|
/// Shared by `execute_statement` and `execute_arrow_statement_prefer_inline`:
|
|
1467
1474
|
/// POST the statement, poll until a terminal state, and turn FAILED/
|
|
1468
1475
|
/// CANCELED into an `Err` -- everything both callers need before they
|
|
@@ -1470,7 +1477,7 @@ impl DbClient {
|
|
|
1470
1477
|
///
|
|
1471
1478
|
/// `body` must *not* already carry `catalog`/`schema`/`session_id` --
|
|
1472
1479
|
/// this method owns that decision: a pooled session for (`catalog`,
|
|
1473
|
-
/// `schema`) if one's available (see `checkout_session
|
|
1480
|
+
/// `schema`) if one's available (see `checkout_session`),
|
|
1474
1481
|
/// falling back to setting `catalog`/`schema` directly on the body
|
|
1475
1482
|
/// otherwise (Databricks rejects `session_id` combined with either
|
|
1476
1483
|
/// field). The session, if any, is returned to the pool on a clean
|
|
@@ -1608,12 +1615,8 @@ impl DbClient {
|
|
|
1608
1615
|
self.host, statement_id, chunk_index
|
|
1609
1616
|
);
|
|
1610
1617
|
let data: ChunkLinksBody = self.authed_json(reqwest::Method::GET, &url, None).await?;
|
|
1611
|
-
|
|
1612
|
-
|
|
1613
|
-
for link in data.external_links {
|
|
1614
|
-
blobs.push(self.fetch_link_bytes(&link.external_link, compressed).await?);
|
|
1615
|
-
}
|
|
1616
|
-
Ok(blobs)
|
|
1618
|
+
let links: Vec<String> = data.external_links.into_iter().map(|l| l.external_link).collect();
|
|
1619
|
+
self.fetch_pre_resolved_links(&links, compressed).await
|
|
1617
1620
|
}
|
|
1618
1621
|
|
|
1619
1622
|
/// Same shape as `fetch_chunk_index`, but for links the statement submit/
|
|
@@ -1668,19 +1671,9 @@ impl DbClient {
|
|
|
1668
1671
|
};
|
|
1669
1672
|
match fetched {
|
|
1670
1673
|
Ok(blobs) => {
|
|
1671
|
-
// `meta.row_count`
|
|
1672
|
-
//
|
|
1673
|
-
// `
|
|
1674
|
-
// over-deliver past its declared count" protection Thrift's
|
|
1675
|
-
// resultLinks and arrowBatches paths both already have, see
|
|
1676
|
-
// AGENTS.md) only when there's exactly one blob, where
|
|
1677
|
-
// "the whole chunk's count" and "this blob's count" are the
|
|
1678
|
-
// same number. A chunk_index resolving to more than one
|
|
1679
|
-
// blob is a real but rare/defensive-coding case (see
|
|
1680
|
-
// `ChunkMeta::pre_resolved_links`'s own doc comment) whose
|
|
1681
|
-
// true per-blob row split isn't known here -- truncating
|
|
1682
|
-
// the first blob to the *whole* chunk's count would be
|
|
1683
|
-
// wrong, so those are left untruncated rather than guessed.
|
|
1674
|
+
// One blob ⇒ `meta.row_count` (the whole chunk_index's
|
|
1675
|
+
// declared count) and "this blob's count" are the same
|
|
1676
|
+
// number -- see `ChunkItem::truncate_to`'s own doc comment.
|
|
1684
1677
|
let truncate_to = if blobs.len() == 1 { meta.row_count } else { None };
|
|
1685
1678
|
for blob in blobs {
|
|
1686
1679
|
let item = ChunkItem {
|
|
@@ -1758,8 +1751,6 @@ impl DbClient {
|
|
|
1758
1751
|
.timeout(self.http_timeout)
|
|
1759
1752
|
.send()
|
|
1760
1753
|
.await
|
|
1761
|
-
// DELETE is naturally idempotent here too (404 is already
|
|
1762
|
-
// treated as success below).
|
|
1763
1754
|
.map_err(|e| ApiError::from_reqwest(e, true))?;
|
|
1764
1755
|
let status = resp.status();
|
|
1765
1756
|
if status == StatusCode::NOT_FOUND || status.is_success() {
|