arrowbricks 3.0.0__tar.gz → 3.0.2__tar.gz

This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
Files changed (35) hide show
  1. {arrowbricks-3.0.0 → arrowbricks-3.0.2}/PKG-INFO +1 -1
  2. {arrowbricks-3.0.0 → arrowbricks-3.0.2}/pyproject.toml +1 -1
  3. {arrowbricks-3.0.0 → arrowbricks-3.0.2}/rust/arrowbricks_core/Cargo.lock +1 -1
  4. {arrowbricks-3.0.0 → arrowbricks-3.0.2}/rust/arrowbricks_core/Cargo.toml +15 -3
  5. {arrowbricks-3.0.0 → arrowbricks-3.0.2}/rust/arrowbricks_core/src/client.rs +180 -168
  6. {arrowbricks-3.0.0 → arrowbricks-3.0.2}/rust/arrowbricks_core/src/heartbeat.rs +20 -42
  7. {arrowbricks-3.0.0 → arrowbricks-3.0.2}/rust/arrowbricks_core/src/json_convert.rs +8 -25
  8. {arrowbricks-3.0.0 → arrowbricks-3.0.2}/rust/arrowbricks_core/src/lib.rs +64 -41
  9. {arrowbricks-3.0.0 → arrowbricks-3.0.2}/rust/arrowbricks_core/src/pipeline.rs +230 -205
  10. {arrowbricks-3.0.0 → arrowbricks-3.0.2}/rust/arrowbricks_core/src/thrift.rs +19 -28
  11. {arrowbricks-3.0.0 → arrowbricks-3.0.2}/rust/arrowbricks_core/tests/wiremock_pipeline.rs +62 -100
  12. {arrowbricks-3.0.0 → arrowbricks-3.0.2}/rust/arrowbricks_core/tests/wiremock_thrift.rs +141 -2
  13. {arrowbricks-3.0.0 → arrowbricks-3.0.2}/rust/arrowbricks_core/tests_py/test_token_provider.py +17 -0
  14. {arrowbricks-3.0.0 → arrowbricks-3.0.2}/rust/arrowbricks_core/tests_py/thrift_mock.py +14 -0
  15. {arrowbricks-3.0.0 → arrowbricks-3.0.2}/src/arrowbricks/_streaming.py +2 -3
  16. {arrowbricks-3.0.0 → arrowbricks-3.0.2}/src/arrowbricks/client.py +9 -25
  17. {arrowbricks-3.0.0 → arrowbricks-3.0.2}/src/arrowbricks/cursor.py +70 -20
  18. {arrowbricks-3.0.0 → arrowbricks-3.0.2}/LICENSE +0 -0
  19. {arrowbricks-3.0.0 → arrowbricks-3.0.2}/README.md +0 -0
  20. {arrowbricks-3.0.0 → arrowbricks-3.0.2}/rust/arrowbricks_core/.gitignore +0 -0
  21. {arrowbricks-3.0.0 → arrowbricks-3.0.2}/rust/arrowbricks_core/README.md +0 -0
  22. {arrowbricks-3.0.0 → arrowbricks-3.0.2}/rust/arrowbricks_core/examples/duckdb_query.py +0 -0
  23. {arrowbricks-3.0.0 → arrowbricks-3.0.2}/rust/arrowbricks_core/examples/fastapi_sse.py +0 -0
  24. {arrowbricks-3.0.0 → arrowbricks-3.0.2}/rust/arrowbricks_core/rustfmt.toml +0 -0
  25. {arrowbricks-3.0.0 → arrowbricks-3.0.2}/rust/arrowbricks_core/tests/wiremock_volume_files.rs +0 -0
  26. {arrowbricks-3.0.0 → arrowbricks-3.0.2}/rust/arrowbricks_core/tests_py/conftest.py +0 -0
  27. {arrowbricks-3.0.0 → arrowbricks-3.0.2}/rust/arrowbricks_core/tests_py/test_ipc_stream.py +0 -0
  28. {arrowbricks-3.0.0 → arrowbricks-3.0.2}/rust/arrowbricks_core/tests_py/test_parameters.py +0 -0
  29. {arrowbricks-3.0.0 → arrowbricks-3.0.2}/rust/arrowbricks_core/tests_py/test_stream_ndjson_lines.py +0 -0
  30. {arrowbricks-3.0.0 → arrowbricks-3.0.2}/rust/arrowbricks_core/tests_py/test_streaming.py +0 -0
  31. {arrowbricks-3.0.0 → arrowbricks-3.0.2}/rust/arrowbricks_core/tests_py/test_thrift.py +0 -0
  32. {arrowbricks-3.0.0 → arrowbricks-3.0.2}/rust/arrowbricks_core/tests_py/test_volume_files.py +0 -0
  33. {arrowbricks-3.0.0 → arrowbricks-3.0.2}/src/arrowbricks/__init__.py +0 -0
  34. {arrowbricks-3.0.0 → arrowbricks-3.0.2}/src/arrowbricks/_core.pyi +0 -0
  35. {arrowbricks-3.0.0 → arrowbricks-3.0.2}/src/arrowbricks/py.typed +0 -0
@@ -1,6 +1,6 @@
1
1
  Metadata-Version: 2.4
2
2
  Name: arrowbricks
3
- Version: 3.0.0
3
+ Version: 3.0.2
4
4
  Requires-Dist: arro3-core>=0.8 ; extra == 'arro3'
5
5
  Provides-Extra: arro3
6
6
  License-File: LICENSE
@@ -1,6 +1,6 @@
1
1
  [project]
2
2
  name = "arrowbricks"
3
- version = "3.0.0"
3
+ version = "3.0.2"
4
4
  description = "Runs SQL against a Databricks SQL warehouse via the Statement Execution API and hands you the result as Arrow -- a DB-API-ish Cursor (fetchone/fetchmany/fetchall/fetchall_arrow) or NDJSON streaming. Rust/PyO3 core throughout -- zero required runtime dependencies."
5
5
  readme = "README.md"
6
6
  license = "MIT"
@@ -240,7 +240,7 @@ dependencies = [
240
240
 
241
241
  [[package]]
242
242
  name = "arrowbricks_core"
243
- version = "3.0.0"
243
+ version = "3.0.2"
244
244
  dependencies = [
245
245
  "arrow",
246
246
  "arrow-json",
@@ -1,6 +1,6 @@
1
1
  [package]
2
2
  name = "arrowbricks_core"
3
- version = "3.0.0"
3
+ version = "3.0.2"
4
4
  edition = "2024"
5
5
  readme = "README.md"
6
6
 
@@ -23,8 +23,13 @@ chrono = { version = "0.4", default-features = false, features = ["std"] }
23
23
  # Decompresses cloud-fetch chunk bytes when the server honors our
24
24
  # `result_compression: LZ4_FRAME` request (see client.rs's execute_statement)
25
25
  # -- the same LZ4 Frame format databricks-sql-python's `lz4.frame` decodes.
26
- # Pure Rust (no C toolchain dependency), matching this crate's reasoning for
27
- # avoiding aws-lc-sys-style native-build pain elsewhere.
26
+ # Pure Rust, no C toolchain dependency -- unlike reqwest's `rustls` feature
27
+ # below, which *does* pull one in (`__rustls-aws-lc-rs` -> aws-lc-rs ->
28
+ # aws-lc-sys, confirmed via `cargo tree -i aws-lc-sys`; this crate has no
29
+ # `ring` feature available in reqwest 0.13 to avoid it with). A previous
30
+ # version of this comment claimed the opposite -- that this crate avoids
31
+ # aws-lc-sys-style native-build pain elsewhere too -- found wrong on review;
32
+ # see AGENTS.md's build-prerequisites note.
28
33
  lz4_flex = { version = "0.11", default-features = false, features = ["frame", "std"] }
29
34
  pyo3 = { version = "0.29.1", features = ["abi3-py311"] }
30
35
  # pyo3-arrow's only optional feature is buffer_protocol (on by default),
@@ -53,6 +58,13 @@ extension-module = ["pyo3/extension-module"]
53
58
  tokio = { version = "1.53.1", features = ["rt-multi-thread", "macros", "test-util", "net", "io-util"] }
54
59
  wiremock = "0.6.5"
55
60
 
61
+ # Currently clean on all three by luck, not enforcement -- pinning them here
62
+ # so a future PR that introduces one fails CI instead of silently landing.
63
+ [lints.clippy]
64
+ redundant_clone = "warn"
65
+ large_enum_variant = "warn"
66
+ needless_collect = "warn"
67
+
56
68
  [profile.release]
57
69
  # opt-level stays at the implicit default (3, speed) rather than "s"/"z" --
58
70
  # perf was the original ask and the benchmarks were run against opt-level 3;
@@ -148,8 +148,7 @@ struct ResultBody {
148
148
  external_links: Vec<ResultLinkBody>,
149
149
  /// Only present for `disposition: INLINE` + `format: JSON_ARRAY` -- each
150
150
  /// row a `Vec<Option<String>>` (every non-null value a string,
151
- /// Databricks' own JSON_ARRAY contract, same as `execute_json_statement`'s
152
- /// normal `EXTERNAL_LINKS`+`JSON_ARRAY` chunks). Consumed by
151
+ /// Databricks' own JSON_ARRAY contract). Consumed by
153
152
  /// `json_convert::json_array_to_record_batch` on the `prefer_inline`
154
153
  /// fast path.
155
154
  #[serde(default)]
@@ -329,7 +328,7 @@ pub struct StatementSubmitResult {
329
328
 
330
329
  /// What `execute_arrow_statement_prefer_inline` gets you -- either the whole
331
330
  /// result inline as raw JSON_ARRAY rows (every non-null value a string,
332
- /// same contract as `execute_json_statement`; converting these into a real
331
+ /// Databricks' own JSON_ARRAY contract; converting these into a real
333
332
  /// `RecordBatch` is `pipeline.rs`'s job via `json_convert`, not this
334
333
  /// Arrow-agnostic module's -- see this file's own module doc comment), or a
335
334
  /// normal `StatementSubmitResult` to fetch chunks for exactly as if
@@ -353,15 +352,18 @@ pub struct ChunkItem {
353
352
  pub blob: Bytes,
354
353
  pub row_count: Option<i64>,
355
354
  pub chunk_index: i64,
356
- /// Set only by the Thrift cloud-fetch path (`pipeline.rs`'s
357
- /// `fetch_thrift_link`) -- a declared row-count bound this chunk's own
358
- /// decoded batches must be sliced down to if they exceed it, since a
359
- /// Thrift `resultLinks` file (like SEA's own cloud-fetch files) can
355
+ /// A declared row-count bound this chunk's own decoded batches must be
356
+ /// sliced down to if they exceed it, since a cloud-fetch file can
360
357
  /// legitimately contain more rows than its own declared count for a
361
358
  /// `LIMIT`-bounded query (see `pipeline.rs`'s `decode_chunk_item`).
362
- /// `None` for every other producer (SEA's own chunk fetch, Thrift's
363
- /// inline `arrowBatches`) -- their row counts are never overshot the
364
- /// same way, so there's nothing to slice.
359
+ /// Every real producer sets this now -- Thrift `resultLinks`
360
+ /// (`pipeline.rs`'s `fetch_thrift_link`), Thrift's inline `arrowBatches`
361
+ /// (`pipeline.rs`'s `run_thrift_fetch_loop`), and SEA's own chunk fetch
362
+ /// (`fetch_chunks_with_backpressure`, below, though only when a
363
+ /// `chunk_index` resolves to exactly one blob -- see that function's own
364
+ /// comment for why more than one blob is left untruncated). `None`
365
+ /// means no authoritative bound is known, not "this chunk was never
366
+ /// overshot" -- `decode_chunk_item` treats it as a no-op either way.
365
367
  pub truncate_to: Option<i64>,
366
368
  }
367
369
 
@@ -507,9 +509,21 @@ pub struct DbClient {
507
509
  /// rebuild the extension to do it. Also doubles, on the Thrift path, as
508
510
  /// `TExecuteStatementReq.canDecompressLZ4Result`.
509
511
  compress_results: bool,
510
- session_pool: SessionPool,
512
+ session_pool: Pool<String>,
511
513
  pub protocol: Protocol,
512
- thrift_session_pool: ThriftSessionPool,
514
+ /// Same checkout/checkin shape as `session_pool` above, plus: Thrift's
515
+ /// session model is *mandatory* per statement, unlike SEA's optional
516
+ /// `session_id`, and this crate has not independently confirmed whether
517
+ /// concurrent statements on one Thrift session are safe against a real
518
+ /// workspace the way the SEA crash was -- so this pool exists
519
+ /// defensively, following the same proven-safe pattern regardless of
520
+ /// whether the analogous crash reproduces here. Unlike SEA, there is no
521
+ /// "sessionless" fallback available at all -- `TExecuteStatementReq.
522
+ /// sessionHandle` is required by the protocol -- so a pool-exhaustion/
523
+ /// creation-failure `None` from `thrift_checkout_session` means the
524
+ /// caller must open one throwaway session for that single call and
525
+ /// close it again immediately after (see `execute_lazy_thrift`).
526
+ thrift_session_pool: Pool<thrift::SessionHandle>,
513
527
  /// Global budget for concurrent cloud-fetch HTTP requests, sized to
514
528
  /// `chunk_fetch_concurrency`. Every link download takes one permit; a
515
529
  /// download that finds spare permits (few links in flight -- exactly
@@ -529,24 +543,12 @@ pub struct DbClient {
529
543
  /// ever split into -- see `DbClient::download_slots`.
530
544
  pub const MAX_SPLIT_PARTS: usize = 8;
531
545
 
532
- /// Session pool for the Thrift backend -- same checkout/checkin shape and
533
- /// the exact same two hard constraints as `SessionPool` above (a session is
534
- /// created *for* one (catalog, schema) pair and can't be redirected; Thrift's
535
- /// session model is *mandatory* per statement, unlike SEA's optional
536
- /// `session_id`, and this crate has not independently confirmed whether
537
- /// concurrent statements on one Thrift session are safe against a real
538
- /// workspace the way the SEA crash was -- so this pool exists defensively,
539
- /// following the same proven-safe pattern regardless of whether the
540
- /// analogous crash reproduces here. Unlike SEA, there is no "sessionless"
541
- /// fallback available at all -- `TExecuteStatementReq.sessionHandle` is
542
- /// required by the protocol -- so a pool-exhaustion/creation-failure `None`
543
- /// here means the caller must open one throwaway session for that single
544
- /// call and close it again immediately after (see `execute_lazy_thrift`).
545
- #[derive(Default)]
546
- struct ThriftSessionPool {
547
- idle: Mutex<HashMap<(Option<String>, Option<String>), Vec<thrift::SessionHandle>>>,
548
- total: Mutex<HashMap<(Option<String>, Option<String>), usize>>,
549
- }
546
+ /// `chunk_fetch_concurrency`'s own default -- see `DbClient::new`'s own
547
+ /// comment on it for the measured numbers behind picking 64. Named so
548
+ /// `download_slots` (sized to this same number, since it's a budget over
549
+ /// the same worker count) can't drift from it the way a second bare `64`
550
+ /// silently could.
551
+ const DEFAULT_CHUNK_FETCH_CONCURRENCY: usize = 64;
550
552
 
551
553
  /// How long the Thrift path polls `GetOperationStatus` when a statement
552
554
  /// doesn't finish within its `getDirectResults` budget (see
@@ -621,13 +623,77 @@ const THRIFT_DIRECT_RESULTS_MAX_BYTES: i64 = 1024 * 1024 * 1024;
621
623
  /// still throws away a perfectly good session), but guarantees a session
622
624
  /// that might be in the same bad state behind the SparkSession-null crash
623
625
  /// above is never handed to a second caller.
624
- #[derive(Default)]
625
- struct SessionPool {
626
+ type SessionKey = (Option<String>, Option<String>);
627
+
628
+ /// Generic checkout/checkin pool keyed by (catalog, schema) -- shared by
629
+ /// SEA's session ids (`Pool<String>`) and Thrift's session handles
630
+ /// (`Pool<thrift::SessionHandle>`), which had this exact logic duplicated
631
+ /// twice over before this. See `checkout_session`/`thrift_checkout_session`
632
+ /// for the constraints that shape it (session pinned to one (catalog,
633
+ /// schema) pair, safe for sequential reuse only, checkout never blocks on
634
+ /// exhaustion).
635
+ struct Pool<T> {
626
636
  // ponytail: fixed cap, not a constructor kwarg -- nothing's asked to
627
637
  // tune this yet; raise (or expose one) if a workload needs more
628
638
  // concurrent sessions per (catalog, schema) pair than this.
629
- idle: Mutex<HashMap<(Option<String>, Option<String>), Vec<String>>>,
630
- total: Mutex<HashMap<(Option<String>, Option<String>), usize>>,
639
+ idle: Mutex<HashMap<SessionKey, Vec<T>>>,
640
+ total: Mutex<HashMap<SessionKey, usize>>,
641
+ }
642
+
643
+ // Manual impl instead of `#[derive(Default)]`: the derive would wrongly
644
+ // require `T: Default` even though neither field actually needs it.
645
+ impl<T> Default for Pool<T> {
646
+ fn default() -> Self {
647
+ Self {
648
+ idle: Mutex::new(HashMap::new()),
649
+ total: Mutex::new(HashMap::new()),
650
+ }
651
+ }
652
+ }
653
+
654
+ impl<T> Pool<T> {
655
+ fn take(&self, key: &SessionKey) -> Option<T> {
656
+ self.idle.lock().unwrap().get_mut(key).and_then(Vec::pop)
657
+ }
658
+
659
+ /// Reserves a slot for `key` if under `MAX_SESSIONS_PER_KEY`. The
660
+ /// increment happens before the caller's own creation call so two
661
+ /// concurrent callers can't both squeeze past the cap; call `release` if
662
+ /// creation then fails.
663
+ fn reserve(&self, key: &SessionKey) -> bool {
664
+ let mut total = self.total.lock().unwrap();
665
+ let count = total.entry(key.clone()).or_insert(0);
666
+ if *count >= MAX_SESSIONS_PER_KEY {
667
+ return false;
668
+ }
669
+ *count += 1;
670
+ true
671
+ }
672
+
673
+ fn release(&self, key: &SessionKey) {
674
+ let mut total = self.total.lock().unwrap();
675
+ if let Some(count) = total.get_mut(key) {
676
+ *count = count.saturating_sub(1);
677
+ }
678
+ }
679
+
680
+ fn put(&self, key: SessionKey, item: T) {
681
+ self.idle.lock().unwrap().entry(key).or_default().push(item);
682
+ }
683
+
684
+ /// Returns `item` to the pool for reuse (`keep = true`) or discards it,
685
+ /// releasing its reservation instead (`keep = false`).
686
+ fn checkin(&self, key: SessionKey, item: T, keep: bool) {
687
+ if keep {
688
+ self.put(key, item);
689
+ } else {
690
+ self.release(&key);
691
+ }
692
+ }
693
+
694
+ fn drain_idle(&self) -> Vec<T> {
695
+ self.idle.lock().unwrap().drain().flat_map(|(_, v)| v).collect()
696
+ }
631
697
  }
632
698
 
633
699
  pub const MAX_SESSIONS_PER_KEY: usize = 8;
@@ -672,15 +738,15 @@ impl DbClient {
672
738
  // a claim that 64 is the true optimum. A caller with a very
673
739
  // large chunk count or a fast/low-latency link to the warehouse
674
740
  // may still want to raise it further.
675
- chunk_fetch_concurrency: 64,
741
+ chunk_fetch_concurrency: DEFAULT_CHUNK_FETCH_CONCURRENCY,
676
742
  warehouse_start_timeout: Duration::from_secs(300),
677
743
  warehouse_confirmed_running_ttl: Duration::from_secs(30),
678
744
  warehouse_confirmed_running_at: Mutex::new(None),
679
745
  compress_results: true,
680
- session_pool: SessionPool::default(),
746
+ session_pool: Pool::default(),
681
747
  protocol: Protocol::Thrift,
682
- thrift_session_pool: ThriftSessionPool::default(),
683
- download_slots: tokio::sync::Semaphore::new(64),
748
+ thrift_session_pool: Pool::default(),
749
+ download_slots: tokio::sync::Semaphore::new(DEFAULT_CHUNK_FETCH_CONCURRENCY),
684
750
  }
685
751
  }
686
752
 
@@ -690,11 +756,7 @@ impl DbClient {
690
756
  self
691
757
  }
692
758
 
693
- /// Selects the wire protocol/backend -- `Protocol::Sea` (this bare
694
- /// `DbClient` constructor's own internal starting value, see
695
- /// `Protocol`'s own doc comment for why that's not the same thing as
696
- /// "the user-facing default") or `Protocol::Thrift` (the actual
697
- /// user-facing default as of this session's benchmarking work).
759
+ /// Selects the wire protocol/backend -- see `Protocol`'s own doc comment.
698
760
  pub fn with_protocol(mut self, protocol: Protocol) -> Self {
699
761
  self.protocol = protocol;
700
762
  self
@@ -830,11 +892,9 @@ impl DbClient {
830
892
  }
831
893
 
832
894
  /// Downloads one cloud-fetch link, splitting it across parallel HTTP
833
- /// Range requests when (and only when) the shared `download_slots`
834
- /// budget has room to spare -- i.e. when few links are in flight, which
835
- /// is exactly the case a single-chunk result hits and where a lone TCP
836
- /// stream leaves most of the link idle. See `download_slots`' own doc
837
- /// comment for the measured win and why the large-result case is safe.
895
+ /// Range requests when the shared `download_slots` budget has room to
896
+ /// spare -- see that field's own doc comment for the measured win and
897
+ /// why the large-result case is safe.
838
898
  pub(crate) async fn fetch_link_bytes_budgeted(
839
899
  self: &Arc<Self>,
840
900
  url: &str,
@@ -843,8 +903,16 @@ impl DbClient {
843
903
  // One permit per download is mandatory; the worker pool already
844
904
  // bounds concurrent links to `chunk_fetch_concurrency`, so this
845
905
  // never blocks in practice -- it just makes the budget accounting
846
- // exact.
847
- let _base = self.download_slots.acquire().await;
906
+ // exact. `acquire()`'s `Err` case (the semaphore closed) can't
907
+ // happen -- nothing in this crate ever calls `.close()` on
908
+ // `download_slots` -- but propagating it as a real error instead of
909
+ // an `.expect()` costs nothing and avoids a panic if that ever
910
+ // changes.
911
+ let _base = self
912
+ .download_slots
913
+ .acquire()
914
+ .await
915
+ .map_err(|e| ApiError::permanent(format!("download_slots semaphore closed unexpectedly: {e}")))?;
848
916
  let want = (MAX_SPLIT_PARTS - 1).min(self.download_slots.available_permits());
849
917
  let extra = if want > 0 {
850
918
  self.download_slots.try_acquire_many(want as u32).ok()
@@ -960,8 +1028,33 @@ impl DbClient {
960
1028
 
961
1029
  let mut out = bytes::BytesMut::with_capacity(total.unwrap_or(head.len() as u64) as usize);
962
1030
  out.extend_from_slice(&head);
963
- for h in handles {
964
- out.extend_from_slice(&h.await.map_err(join_error)??);
1031
+ // Parts must concatenate in order, so this can't use `JoinSet`
1032
+ // (which yields in completion order) without tracking indices --
1033
+ // simpler to keep the `Vec` and explicitly `.abort()` every
1034
+ // not-yet-awaited sibling the moment one part fails, rather than
1035
+ // silently leaving them running (a bare `?` here would return
1036
+ // early and just drop the rest, which does NOT cancel them --
1037
+ // `JoinHandle::drop` detaches, it doesn't abort -- leaving up to
1038
+ // `MAX_SPLIT_PARTS - 1` sibling Range downloads, each with their
1039
+ // own `retry_call` backoff, still in flight for a link the caller
1040
+ // has already given up on).
1041
+ let mut iter = handles.into_iter();
1042
+ while let Some(h) = iter.next() {
1043
+ match h.await {
1044
+ Ok(Ok(part)) => out.extend_from_slice(&part),
1045
+ Ok(Err(e)) => {
1046
+ for remaining in iter {
1047
+ remaining.abort();
1048
+ }
1049
+ return Err(e);
1050
+ }
1051
+ Err(join_err) => {
1052
+ for remaining in iter {
1053
+ remaining.abort();
1054
+ }
1055
+ return Err(join_error(join_err));
1056
+ }
1057
+ }
965
1058
  }
966
1059
  let bytes = out.freeze();
967
1060
  if let Some(t) = total
@@ -980,13 +1073,16 @@ impl DbClient {
980
1073
  .map_err(join_error)?
981
1074
  }
982
1075
 
983
- async fn ensure_warehouse_running(&self) -> Result<(), ApiError> {
1076
+ /// Shared by both protocols -- plain REST against `/api/2.0/sql/warehouses/{id}`,
1077
+ /// nothing SEA- or Thrift-specific about it (see AGENTS.md for why the
1078
+ /// Thrift path didn't always call this).
1079
+ pub(crate) async fn ensure_warehouse_running(&self) -> Result<(), ApiError> {
984
1080
  {
985
1081
  let confirmed = *self.warehouse_confirmed_running_at.lock().unwrap();
986
- if let Some(at) = confirmed {
987
- if at.elapsed() < self.warehouse_confirmed_running_ttl {
988
- return Ok(());
989
- }
1082
+ if let Some(at) = confirmed
1083
+ && at.elapsed() < self.warehouse_confirmed_running_ttl
1084
+ {
1085
+ return Ok(());
990
1086
  }
991
1087
  }
992
1088
 
@@ -1041,36 +1137,20 @@ impl DbClient {
1041
1137
  /// Hands back a pooled session for (`catalog`, `schema`) if one's idle,
1042
1138
  /// creates one if the pool for that key isn't at `MAX_SESSIONS_PER_KEY`
1043
1139
  /// yet, or `None` if neither -- the caller falls back to a plain
1044
- /// session-less submission in that case, see `SessionPool`'s own doc
1045
- /// comment for why this never blocks instead. The slot-reservation
1046
- /// increment happens *before* the `create_session` await so two
1047
- /// concurrent callers can't both squeeze past the cap; a failed creation
1048
- /// releases the reservation again.
1140
+ /// session-less submission in that case, see `session_pool`'s own doc
1141
+ /// comment for why this never blocks instead.
1049
1142
  async fn checkout_session(&self, catalog: Option<&str>, schema: Option<&str>) -> Option<String> {
1050
1143
  let key = (catalog.map(str::to_string), schema.map(str::to_string));
1051
- {
1052
- let mut idle = self.session_pool.idle.lock().unwrap();
1053
- if let Some(ids) = idle.get_mut(&key)
1054
- && let Some(id) = ids.pop()
1055
- {
1056
- return Some(id);
1057
- }
1144
+ if let Some(id) = self.session_pool.take(&key) {
1145
+ return Some(id);
1058
1146
  }
1059
- {
1060
- let mut total = self.session_pool.total.lock().unwrap();
1061
- let count = total.entry(key.clone()).or_insert(0);
1062
- if *count >= MAX_SESSIONS_PER_KEY {
1063
- return None;
1064
- }
1065
- *count += 1;
1147
+ if !self.session_pool.reserve(&key) {
1148
+ return None;
1066
1149
  }
1067
1150
  match self.create_session(catalog, schema).await {
1068
1151
  Ok(id) => Some(id),
1069
1152
  Err(_) => {
1070
- let mut total = self.session_pool.total.lock().unwrap();
1071
- if let Some(count) = total.get_mut(&key) {
1072
- *count = count.saturating_sub(1);
1073
- }
1153
+ self.session_pool.release(&key);
1074
1154
  None
1075
1155
  }
1076
1156
  }
@@ -1078,24 +1158,11 @@ impl DbClient {
1078
1158
 
1079
1159
  /// Returns a session to the pool for reuse (`keep = true`, the statement
1080
1160
  /// it backed reached SUCCEEDED/FAILED/CANCELED cleanly) or discards it
1081
- /// (`keep = false`) -- see `SessionPool`'s doc comment for why any error
1161
+ /// (`keep = false`) -- see `session_pool`'s doc comment for why any error
1082
1162
  /// discards rather than reuses.
1083
1163
  fn checkin_session(&self, catalog: Option<&str>, schema: Option<&str>, session_id: String, keep: bool) {
1084
1164
  let key = (catalog.map(str::to_string), schema.map(str::to_string));
1085
- if keep {
1086
- self.session_pool
1087
- .idle
1088
- .lock()
1089
- .unwrap()
1090
- .entry(key)
1091
- .or_default()
1092
- .push(session_id);
1093
- } else {
1094
- let mut total = self.session_pool.total.lock().unwrap();
1095
- if let Some(count) = total.get_mut(&key) {
1096
- *count = count.saturating_sub(1);
1097
- }
1098
- }
1165
+ self.session_pool.checkin(key, session_id, keep);
1099
1166
  }
1100
1167
 
1101
1168
  /// Best-effort cleanup of every currently-idle pooled session -- meant
@@ -1106,11 +1173,7 @@ impl DbClient {
1106
1173
  /// this before every pending statement has finished is a caller
1107
1174
  /// ordering issue, not something this method can fix from inside.
1108
1175
  pub async fn close_all_sessions(&self) {
1109
- let ids: Vec<String> = {
1110
- let mut idle = self.session_pool.idle.lock().unwrap();
1111
- idle.drain().flat_map(|(_, v)| v).collect()
1112
- };
1113
- for id in ids {
1176
+ for id in self.session_pool.drain_idle() {
1114
1177
  self.delete_session(&id).await;
1115
1178
  }
1116
1179
  }
@@ -1201,8 +1264,8 @@ impl DbClient {
1201
1264
  }
1202
1265
 
1203
1266
  /// Same checkout contract as SEA's `checkout_session` (see
1204
- /// `ThriftSessionPool`'s doc comment): `None` means the caller must open
1205
- /// its own throwaway session for this one call (Thrift has no
1267
+ /// `thrift_session_pool`'s doc comment): `None` means the caller must
1268
+ /// open its own throwaway session for this one call (Thrift has no
1206
1269
  /// session-less submission mode to fall back to).
1207
1270
  pub(crate) async fn thrift_checkout_session(
1208
1271
  &self,
@@ -1210,29 +1273,16 @@ impl DbClient {
1210
1273
  schema: Option<&str>,
1211
1274
  ) -> Option<thrift::SessionHandle> {
1212
1275
  let key = (catalog.map(str::to_string), schema.map(str::to_string));
1213
- {
1214
- let mut idle = self.thrift_session_pool.idle.lock().unwrap();
1215
- if let Some(ids) = idle.get_mut(&key)
1216
- && let Some(id) = ids.pop()
1217
- {
1218
- return Some(id);
1219
- }
1276
+ if let Some(handle) = self.thrift_session_pool.take(&key) {
1277
+ return Some(handle);
1220
1278
  }
1221
- {
1222
- let mut total = self.thrift_session_pool.total.lock().unwrap();
1223
- let count = total.entry(key.clone()).or_insert(0);
1224
- if *count >= MAX_SESSIONS_PER_KEY {
1225
- return None;
1226
- }
1227
- *count += 1;
1279
+ if !self.thrift_session_pool.reserve(&key) {
1280
+ return None;
1228
1281
  }
1229
1282
  match self.thrift_open_session_raw(catalog, schema).await {
1230
1283
  Ok(handle) => Some(handle),
1231
1284
  Err(_) => {
1232
- let mut total = self.thrift_session_pool.total.lock().unwrap();
1233
- if let Some(count) = total.get_mut(&key) {
1234
- *count = count.saturating_sub(1);
1235
- }
1285
+ self.thrift_session_pool.release(&key);
1236
1286
  None
1237
1287
  }
1238
1288
  }
@@ -1246,30 +1296,13 @@ impl DbClient {
1246
1296
  keep: bool,
1247
1297
  ) {
1248
1298
  let key = (catalog.map(str::to_string), schema.map(str::to_string));
1249
- if keep {
1250
- self.thrift_session_pool
1251
- .idle
1252
- .lock()
1253
- .unwrap()
1254
- .entry(key)
1255
- .or_default()
1256
- .push(session);
1257
- } else {
1258
- let mut total = self.thrift_session_pool.total.lock().unwrap();
1259
- if let Some(count) = total.get_mut(&key) {
1260
- *count = count.saturating_sub(1);
1261
- }
1262
- }
1299
+ self.thrift_session_pool.checkin(key, session, keep);
1263
1300
  }
1264
1301
 
1265
1302
  /// Best-effort close of every currently-idle pooled Thrift session --
1266
1303
  /// same contract as `close_all_sessions` (SEA).
1267
1304
  pub async fn close_all_thrift_sessions(&self) {
1268
- let sessions: Vec<thrift::SessionHandle> = {
1269
- let mut idle = self.thrift_session_pool.idle.lock().unwrap();
1270
- idle.drain().flat_map(|(_, v)| v).collect()
1271
- };
1272
- for s in sessions {
1305
+ for s in self.thrift_session_pool.drain_idle() {
1273
1306
  self.thrift_close_session_raw(&s).await;
1274
1307
  }
1275
1308
  }
@@ -1305,7 +1338,6 @@ impl DbClient {
1305
1338
  op: &thrift::OperationHandle,
1306
1339
  ) -> Result<thrift::OperationStatusResp, ApiError> {
1307
1340
  let body = Bytes::from(thrift::build_get_operation_status(op));
1308
- // Idempotent -- read-only.
1309
1341
  let resp_bytes = self.thrift_call(body, true).await?;
1310
1342
  thrift::parse_get_operation_status(&resp_bytes).map_err(Self::thrift_parse_error)
1311
1343
  }
@@ -1438,24 +1470,6 @@ impl DbClient {
1438
1470
  }
1439
1471
  }
1440
1472
 
1441
- /// Like `execute_arrow_statement`, fixed to JSON_ARRAY -- each fetched
1442
- /// chunk's bytes are then a JSON array of rows, each row itself an array
1443
- /// of values where every non-null value is a *string* regardless of its
1444
- /// real column type (Databricks' own JSON_ARRAY contract; null stays
1445
- /// JSON null) -- casting by the manifest's column type_name, if wanted,
1446
- /// is left to the caller, same as the Python original does nothing extra
1447
- /// here either.
1448
- pub async fn execute_json_statement(
1449
- &self,
1450
- statement: &str,
1451
- catalog: Option<&str>,
1452
- schema: Option<&str>,
1453
- parameters: Option<Value>,
1454
- ) -> Result<StatementSubmitResult, ApiError> {
1455
- self.execute_statement(statement, "JSON_ARRAY", catalog, schema, parameters)
1456
- .await
1457
- }
1458
-
1459
1473
  /// Shared by `execute_statement` and `execute_arrow_statement_prefer_inline`:
1460
1474
  /// POST the statement, poll until a terminal state, and turn FAILED/
1461
1475
  /// CANCELED into an `Err` -- everything both callers need before they
@@ -1463,7 +1477,7 @@ impl DbClient {
1463
1477
  ///
1464
1478
  /// `body` must *not* already carry `catalog`/`schema`/`session_id` --
1465
1479
  /// this method owns that decision: a pooled session for (`catalog`,
1466
- /// `schema`) if one's available (see `checkout_session`/`SessionPool`),
1480
+ /// `schema`) if one's available (see `checkout_session`),
1467
1481
  /// falling back to setting `catalog`/`schema` directly on the body
1468
1482
  /// otherwise (Databricks rejects `session_id` combined with either
1469
1483
  /// field). The session, if any, is returned to the pool on a clean
@@ -1601,12 +1615,8 @@ impl DbClient {
1601
1615
  self.host, statement_id, chunk_index
1602
1616
  );
1603
1617
  let data: ChunkLinksBody = self.authed_json(reqwest::Method::GET, &url, None).await?;
1604
-
1605
- let mut blobs = Vec::with_capacity(data.external_links.len());
1606
- for link in data.external_links {
1607
- blobs.push(self.fetch_link_bytes(&link.external_link, compressed).await?);
1608
- }
1609
- Ok(blobs)
1618
+ let links: Vec<String> = data.external_links.into_iter().map(|l| l.external_link).collect();
1619
+ self.fetch_pre_resolved_links(&links, compressed).await
1610
1620
  }
1611
1621
 
1612
1622
  /// Same shape as `fetch_chunk_index`, but for links the statement submit/
@@ -1661,12 +1671,16 @@ impl DbClient {
1661
1671
  };
1662
1672
  match fetched {
1663
1673
  Ok(blobs) => {
1674
+ // One blob ⇒ `meta.row_count` (the whole chunk_index's
1675
+ // declared count) and "this blob's count" are the same
1676
+ // number -- see `ChunkItem::truncate_to`'s own doc comment.
1677
+ let truncate_to = if blobs.len() == 1 { meta.row_count } else { None };
1664
1678
  for blob in blobs {
1665
1679
  let item = ChunkItem {
1666
1680
  blob,
1667
1681
  row_count: meta.row_count,
1668
1682
  chunk_index: meta.chunk_index,
1669
- truncate_to: None,
1683
+ truncate_to,
1670
1684
  };
1671
1685
  if worker_tx.send(Ok(item)).await.is_err() {
1672
1686
  return Ok(());
@@ -1737,8 +1751,6 @@ impl DbClient {
1737
1751
  .timeout(self.http_timeout)
1738
1752
  .send()
1739
1753
  .await
1740
- // DELETE is naturally idempotent here too (404 is already
1741
- // treated as success below).
1742
1754
  .map_err(|e| ApiError::from_reqwest(e, true))?;
1743
1755
  let status = resp.status();
1744
1756
  if status == StatusCode::NOT_FOUND || status.is_success() {