arrowbricks 1.3.2__tar.gz → 1.4.0__tar.gz

This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
Files changed (31) hide show
  1. {arrowbricks-1.3.2 → arrowbricks-1.4.0}/PKG-INFO +3 -3
  2. {arrowbricks-1.3.2 → arrowbricks-1.4.0}/README.md +2 -2
  3. {arrowbricks-1.3.2 → arrowbricks-1.4.0}/pyproject.toml +3 -3
  4. {arrowbricks-1.3.2 → arrowbricks-1.4.0}/rust/arrowbricks_core/Cargo.lock +2 -1
  5. {arrowbricks-1.3.2 → arrowbricks-1.4.0}/rust/arrowbricks_core/Cargo.toml +6 -1
  6. {arrowbricks-1.3.2 → arrowbricks-1.4.0}/rust/arrowbricks_core/README.md +6 -7
  7. {arrowbricks-1.3.2 → arrowbricks-1.4.0}/rust/arrowbricks_core/examples/duckdb_query.py +2 -1
  8. {arrowbricks-1.3.2 → arrowbricks-1.4.0}/rust/arrowbricks_core/src/client.rs +244 -41
  9. arrowbricks-1.4.0/rust/arrowbricks_core/src/json_convert.rs +476 -0
  10. {arrowbricks-1.3.2 → arrowbricks-1.4.0}/rust/arrowbricks_core/src/lib.rs +66 -154
  11. {arrowbricks-1.3.2 → arrowbricks-1.4.0}/rust/arrowbricks_core/src/pipeline.rs +178 -3
  12. {arrowbricks-1.3.2 → arrowbricks-1.4.0}/rust/arrowbricks_core/tests/wiremock_pipeline.rs +354 -1
  13. {arrowbricks-1.3.2 → arrowbricks-1.4.0}/rust/arrowbricks_core/tests_py/test_parameters.py +3 -16
  14. {arrowbricks-1.3.2 → arrowbricks-1.4.0}/rust/arrowbricks_core/tests_py/test_streaming.py +42 -33
  15. {arrowbricks-1.3.2 → arrowbricks-1.4.0}/rust/arrowbricks_core/tests_py/test_token_provider.py +52 -2
  16. {arrowbricks-1.3.2 → arrowbricks-1.4.0}/src/arrowbricks/_core.pyi +1 -27
  17. {arrowbricks-1.3.2 → arrowbricks-1.4.0}/src/arrowbricks/client.py +18 -18
  18. {arrowbricks-1.3.2 → arrowbricks-1.4.0}/src/arrowbricks/cursor.py +33 -3
  19. arrowbricks-1.3.2/rust/arrowbricks_core/tests_py/test_execute_json.py +0 -101
  20. {arrowbricks-1.3.2 → arrowbricks-1.4.0}/LICENSE +0 -0
  21. {arrowbricks-1.3.2 → arrowbricks-1.4.0}/rust/arrowbricks_core/.gitignore +0 -0
  22. {arrowbricks-1.3.2 → arrowbricks-1.4.0}/rust/arrowbricks_core/examples/fastapi_sse.py +0 -0
  23. {arrowbricks-1.3.2 → arrowbricks-1.4.0}/rust/arrowbricks_core/rustfmt.toml +0 -0
  24. {arrowbricks-1.3.2 → arrowbricks-1.4.0}/rust/arrowbricks_core/src/heartbeat.rs +0 -0
  25. {arrowbricks-1.3.2 → arrowbricks-1.4.0}/rust/arrowbricks_core/tests/wiremock_volume_files.rs +0 -0
  26. {arrowbricks-1.3.2 → arrowbricks-1.4.0}/rust/arrowbricks_core/tests_py/test_ipc_stream.py +0 -0
  27. {arrowbricks-1.3.2 → arrowbricks-1.4.0}/rust/arrowbricks_core/tests_py/test_stream_ndjson_lines.py +0 -0
  28. {arrowbricks-1.3.2 → arrowbricks-1.4.0}/rust/arrowbricks_core/tests_py/test_volume_files.py +0 -0
  29. {arrowbricks-1.3.2 → arrowbricks-1.4.0}/src/arrowbricks/__init__.py +0 -0
  30. {arrowbricks-1.3.2 → arrowbricks-1.4.0}/src/arrowbricks/_streaming.py +0 -0
  31. {arrowbricks-1.3.2 → arrowbricks-1.4.0}/src/arrowbricks/py.typed +0 -0
@@ -1,6 +1,6 @@
1
1
  Metadata-Version: 2.4
2
2
  Name: arrowbricks
3
- Version: 1.3.2
3
+ Version: 1.4.0
4
4
  Requires-Dist: arro3-core>=0.8 ; extra == 'arro3'
5
5
  Provides-Extra: arro3
6
6
  License-File: LICENSE
@@ -139,7 +139,7 @@ conn = connect(host=..., warehouse_id=..., token_provider=my_token_provider)
139
139
  - `connect(host, warehouse_id, *, token=None, token_provider=None, ...) -> Connection`
140
140
  - `Connection.cursor() -> Cursor`
141
141
  - `Connection.client -> DatabricksClient` -- the same client `cursor()` uses, for lower-level access (e.g. `stream_query_json`, `upload_volume_file`).
142
- - `Cursor.execute(sql, parameters=None, *, row_limit=None, offset=None, catalog=None, schema=None, total_timeout_s=None) -> Cursor` -- submits and waits for the statement, like a real DB-API cursor. `parameters`, if given, is Databricks' own named-parameter format -- `[{"name": ..., "value": ..., "type": ...}]` bound against `:name` markers in `sql`.
142
+ - `Cursor.execute(sql, parameters=None, *, row_limit=None, offset=None, catalog=None, schema=None, total_timeout_s=None, prefer_inline=False) -> Cursor` -- submits and waits for the statement, like a real DB-API cursor. `parameters`, if given, is Databricks' own named-parameter format -- `[{"name": ..., "value": ..., "type": ...}]` bound against `:name` markers in `sql`. `prefer_inline=True` tries fetching a small result (well under Databricks' 25 MiB inline cap) in the same round trip as the submission itself, skipping the chunk-fetch entirely -- if the result turns out too big, or has a column type this can't convert (nested ARRAY/MAP/STRUCT, VARIANT), it transparently re-runs the query the normal way, so a caller who sets this without actually expecting a small result pays for the query twice. Leave it off unless you know the result is small.
143
143
  - `Cursor.execute_streamed(...)` -- same args, but an async generator yielding `HEARTBEAT` while waiting on a slow cold start, then the ready `Cursor` -- for bridging e.g. an SSE connection. Its timeout/heartbeats stop the moment the statement is ready, *before* any chunk has been downloaded -- see `fetchall_streamed` below for the download phase itself.
144
144
  - `Cursor.fetchone() -> tuple | None`, `Cursor.fetchmany(size) -> list[tuple]`, `Cursor.fetchall() -> list[tuple]`, and iterating a `Cursor` directly -- row tuples; needs the `arro3` extra.
145
145
  - `Cursor.fetchmany_arrow(size) -> Table`, `Cursor.fetchall_arrow() -> Table` -- an Arrow table (implements `__arrow_c_stream__`, so arro3/pyarrow/DuckDB can all consume it directly, zero-copy).
@@ -191,7 +191,7 @@ con.register("my_table", table)
191
191
  con.sql("SELECT count(*) FROM my_table").show()
192
192
  ```
193
193
 
194
- `ReplayableArrowChunk` works the same way for Arrow-IPC bytes you fetched and stored earlier (e.g. `client.execute_arrow_statement(...)`'s raw chunk bytes, cached in Redis/a file/wherever) -- DuckDB's registration path calls `__arrow_c_stream__` twice (a schema peek, then the actual scan), which is exactly what `ReplayableArrowChunk` exists to support:
194
+ `ReplayableArrowChunk` works the same way for Arrow-IPC bytes you fetched and stored earlier (e.g. a raw chunk's bytes, cached in Redis/a file/wherever) -- DuckDB's registration path calls `__arrow_c_stream__` twice (a schema peek, then the actual scan), which is exactly what `ReplayableArrowChunk` exists to support:
195
195
 
196
196
  ```python
197
197
  from arrowbricks import ReplayableArrowChunk
@@ -127,7 +127,7 @@ conn = connect(host=..., warehouse_id=..., token_provider=my_token_provider)
127
127
  - `connect(host, warehouse_id, *, token=None, token_provider=None, ...) -> Connection`
128
128
  - `Connection.cursor() -> Cursor`
129
129
  - `Connection.client -> DatabricksClient` -- the same client `cursor()` uses, for lower-level access (e.g. `stream_query_json`, `upload_volume_file`).
130
- - `Cursor.execute(sql, parameters=None, *, row_limit=None, offset=None, catalog=None, schema=None, total_timeout_s=None) -> Cursor` -- submits and waits for the statement, like a real DB-API cursor. `parameters`, if given, is Databricks' own named-parameter format -- `[{"name": ..., "value": ..., "type": ...}]` bound against `:name` markers in `sql`.
130
+ - `Cursor.execute(sql, parameters=None, *, row_limit=None, offset=None, catalog=None, schema=None, total_timeout_s=None, prefer_inline=False) -> Cursor` -- submits and waits for the statement, like a real DB-API cursor. `parameters`, if given, is Databricks' own named-parameter format -- `[{"name": ..., "value": ..., "type": ...}]` bound against `:name` markers in `sql`. `prefer_inline=True` tries fetching a small result (well under Databricks' 25 MiB inline cap) in the same round trip as the submission itself, skipping the chunk-fetch entirely -- if the result turns out too big, or has a column type this can't convert (nested ARRAY/MAP/STRUCT, VARIANT), it transparently re-runs the query the normal way, so a caller who sets this without actually expecting a small result pays for the query twice. Leave it off unless you know the result is small.
131
131
  - `Cursor.execute_streamed(...)` -- same args, but an async generator yielding `HEARTBEAT` while waiting on a slow cold start, then the ready `Cursor` -- for bridging e.g. an SSE connection. Its timeout/heartbeats stop the moment the statement is ready, *before* any chunk has been downloaded -- see `fetchall_streamed` below for the download phase itself.
132
132
  - `Cursor.fetchone() -> tuple | None`, `Cursor.fetchmany(size) -> list[tuple]`, `Cursor.fetchall() -> list[tuple]`, and iterating a `Cursor` directly -- row tuples; needs the `arro3` extra.
133
133
  - `Cursor.fetchmany_arrow(size) -> Table`, `Cursor.fetchall_arrow() -> Table` -- an Arrow table (implements `__arrow_c_stream__`, so arro3/pyarrow/DuckDB can all consume it directly, zero-copy).
@@ -179,7 +179,7 @@ con.register("my_table", table)
179
179
  con.sql("SELECT count(*) FROM my_table").show()
180
180
  ```
181
181
 
182
- `ReplayableArrowChunk` works the same way for Arrow-IPC bytes you fetched and stored earlier (e.g. `client.execute_arrow_statement(...)`'s raw chunk bytes, cached in Redis/a file/wherever) -- DuckDB's registration path calls `__arrow_c_stream__` twice (a schema peek, then the actual scan), which is exactly what `ReplayableArrowChunk` exists to support:
182
+ `ReplayableArrowChunk` works the same way for Arrow-IPC bytes you fetched and stored earlier (e.g. a raw chunk's bytes, cached in Redis/a file/wherever) -- DuckDB's registration path calls `__arrow_c_stream__` twice (a schema peek, then the actual scan), which is exactly what `ReplayableArrowChunk` exists to support:
183
183
 
184
184
  ```python
185
185
  from arrowbricks import ReplayableArrowChunk
@@ -1,6 +1,6 @@
1
1
  [project]
2
2
  name = "arrowbricks"
3
- version = "1.3.2"
3
+ version = "1.4.0"
4
4
  description = "Runs SQL against a Databricks SQL warehouse via the Statement Execution API and hands you the result as Arrow -- a DB-API-ish Cursor (fetchone/fetchmany/fetchall/fetchall_arrow) or NDJSON streaming. Rust/PyO3 core throughout -- zero required runtime dependencies."
5
5
  readme = "README.md"
6
6
  license = "MIT"
@@ -10,8 +10,8 @@ requires-python = ">=3.11"
10
10
  # (see rust/arrowbricks_core's own reqwest client + Arrow-IPC/NDJSON
11
11
  # writer). arro3-core is optional (see below): only Cursor.fetchone/
12
12
  # fetchmany/fetchall's row materialization needs it -- fetchall_arrow,
13
- # execute_arrow, stream_query_json, upload_volume_file/delete_volume_file,
14
- # and Cursor.description all work with zero dependencies installed.
13
+ # stream_query_json, upload_volume_file/delete_volume_file, and
14
+ # Cursor.description all work with zero dependencies installed.
15
15
  dependencies = []
16
16
 
17
17
  [project.optional-dependencies]
@@ -240,11 +240,12 @@ dependencies = [
240
240
 
241
241
  [[package]]
242
242
  name = "arrowbricks_core"
243
- version = "1.3.2"
243
+ version = "1.4.0"
244
244
  dependencies = [
245
245
  "arrow",
246
246
  "arrow-json",
247
247
  "bytes",
248
+ "chrono",
248
249
  "lz4_flex",
249
250
  "pyo3",
250
251
  "pyo3-arrow",
@@ -1,6 +1,6 @@
1
1
  [package]
2
2
  name = "arrowbricks_core"
3
- version = "1.3.2"
3
+ version = "1.4.0"
4
4
  edition = "2024"
5
5
  readme = "README.md"
6
6
 
@@ -15,6 +15,11 @@ crate-type = ["cdylib", "rlib"]
15
15
  arrow = { version = "59.1.0", default-features = false, features = ["ipc"] }
16
16
  arrow-json = "59.1.0"
17
17
  bytes = "1.12.1"
18
+ # Already an unconditional transitive dependency of pyo3-arrow (see its own
19
+ # comment below) -- declared directly here too so json_convert.rs can parse
20
+ # DATE/TIMESTAMP/TIMESTAMP_NTZ strings without hand-rolling calendar
21
+ # arithmetic. No new code ships that wasn't already linked in.
22
+ chrono = { version = "0.4", default-features = false, features = ["std"] }
18
23
  # Decompresses cloud-fetch chunk bytes when the server honors our
19
24
  # `result_compression: LZ4_FRAME` request (see client.rs's execute_statement)
20
25
  # -- the same LZ4 Frame format databricks-sql-python's `lz4.frame` decodes.
@@ -48,7 +48,8 @@ async def main():
48
48
  warehouse_id="abcd1234efgh5678",
49
49
  token="dapi...",
50
50
  )
51
- table = await client.execute_arrow("SELECT * FROM my_catalog.my_schema.my_table LIMIT 100")
51
+ result = await client.execute("SELECT * FROM my_catalog.my_schema.my_table LIMIT 100")
52
+ table = await result.fetchall_arrow()
52
53
  print(table.num_rows)
53
54
 
54
55
 
@@ -66,17 +67,14 @@ unlike the table object itself.
66
67
  ## API
67
68
 
68
69
  - `Client(host, warehouse_id, *, token=None, token_provider=None, chunk_fetch_concurrency=64, http_timeout=60.0, wait_timeout="30s", warehouse_start_timeout=300.0, warehouse_confirmed_running_ttl_s=30.0, compress_results=True)` -- exactly one of `token`/`token_provider`. `token_provider` is a callable (sync or async) returning a token string, called fresh on every request, no caching. `compress_results` requests LZ4-compressed cloud-fetch chunks (see above); set `False` to opt out.
69
- - `Client.execute_arrow(statement, *, catalog=None, schema=None, parameters=None) -> Table` -- eager: fetches and assembles the whole result before returning. `parameters` is Databricks' own named-parameter format (`[{"name":..., "value":..., "type":...}]`), passed straight through.
70
- - `Client.execute(statement, *, catalog=None, schema=None, parameters=None) -> ResultSet` -- submits and starts background chunk fetching without pulling anything yet.
70
+ - `Client.execute(statement, *, catalog=None, schema=None, parameters=None, prefer_inline=False) -> ResultSet` -- submits and starts background chunk fetching without pulling anything yet. `parameters` is Databricks' own named-parameter format (`[{"name":..., "value":..., "type":...}]`), passed straight through. `prefer_inline=True` submits with `disposition=INLINE, format=JSON_ARRAY` instead, for a caller who expects a small (well under Databricks' 25 MiB inline cap) result and wants to skip the chunk-fetch round trip -- on a result too big for INLINE, or containing a column type `json_convert.rs` doesn't map (STRUCT/ARRAY-of-STRUCT/MAP/VARIANT), it transparently falls back to a second, normal `execute()` and the query runs twice. See `AGENTS.md`'s "Design invariants" section (in the root package) for the full reasoning and real-workspace verification behind this.
71
71
  - `ResultSet.fetchmany_arrow(n) -> Table` -- pulls/decodes only as many chunks as needed for `n` rows, buffering the rest; may return fewer than `n` once exhausted.
72
72
  - `ResultSet.fetchall_arrow() -> Table` -- drains everything remaining.
73
+ - `ResultSet.fetchall_arrow_streamed(*, total_timeout_s=None)` -- same as `fetchall_arrow()`, but an async iterator yielding the `HEARTBEAT` singleton while pulling chunks instead of blocking silently (bridge e.g. an SSE connection through the download), then a `Table` exactly once. Raises if `total_timeout_s` elapses first.
73
74
  - `ResultSet.schema() -> list[tuple[str, str]] | None` -- the real decoded Arrow schema as `(name, type_name)` string pairs, once known (after >=1 fetch); `None` before that. Computed directly in Rust -- unlike a `Table`'s own `.schema`, this needs no arro3 install.
74
75
  - `ResultSet.statement_id`, `ResultSet.num_chunks`, `ResultSet.columns` (manifest-based pre-fetch schema estimate, also arro3-free).
75
- - `Client.execute_json(statement, *, catalog=None, schema=None, parameters=None) -> list[list[str | None]]` -- JSON_ARRAY format, no Arrow parse. Every non-null value comes back as a string regardless of real column type (Databricks' own contract) -- cast by the manifest's column type yourself if you want native Python types.
76
76
  - `Client.stream_ndjson_lines(statement, *, catalog=None, schema=None, parameters=None, total_timeout_s=None)` -- chunk-at-a-time: an async iterator yielding the `HEARTBEAT` singleton while waiting on the statement or any individual chunk, then a `list[str]` of NDJSON lines (one per row, explicit nulls, ISO-8601 timestamps) per chunk in logical order. Decode and JSON encoding both happen in Rust -- backs `stream_query_json` end to end.
77
77
  - `Client.upload_volume_file(volume_path, data: bytes)` / `Client.delete_volume_file(volume_path)` -- Unity Catalog volume files via the Files API. Delete treats a 404 as success (idempotent). Both raise a plain `RuntimeError` (message only) on failure.
78
- - `Client.execute_streamed(statement, *, catalog=None, schema=None, parameters=None, total_timeout_s=None)` -- like `execute()`, but an async iterator yielding the `HEARTBEAT` singleton while waiting on Databricks instead of blocking silently (bridge e.g. an SSE connection through a slow warehouse cold start), then a `ResultSet` exactly once. Raises if `total_timeout_s` elapses first.
79
- - `ResultSet.fetchall_arrow_streamed(*, total_timeout_s=None)` -- same idea for the chunk-download phase: yields `HEARTBEAT` while pulling chunks, then a `Table` exactly once.
80
78
  - `write_ipc_stream(stream, buf)` -- free function; writes any object implementing `__arrow_c_stream__` (a `Table` from this crate, arro3, pyarrow, ...) as uncompressed Arrow-IPC stream bytes to a Python file-like object. No dependency needed regardless of the input's origin.
81
79
  - `read_ipc_stream(data: bytes) -> Table` -- free function; the exact inverse of `write_ipc_stream`, parsing raw Arrow-IPC stream bytes back into a `Table`. No dependency needed regardless of where the bytes came from -- backs `arrowbricks.ReplayableArrowChunk`, which needs to re-parse the same cached bytes on every `__arrow_c_stream__` call.
82
80
  - `HEARTBEAT` -- module-level singleton; compare with `is`, e.g. `if item is _core.HEARTBEAT: ...`.
@@ -90,7 +88,8 @@ directly, no pandas/pyarrow conversion step, no extra dependency:
90
88
  ```python
91
89
  import duckdb
92
90
 
93
- result = await client.execute_arrow("SELECT * FROM my_catalog.my_schema.my_table")
91
+ result_set = await client.execute("SELECT * FROM my_catalog.my_schema.my_table")
92
+ result = await result_set.fetchall_arrow()
94
93
  print(duckdb.sql("SELECT count(*) FROM result").fetchall())
95
94
  ```
96
95
 
@@ -27,7 +27,8 @@ async def main() -> None:
27
27
  warehouse_id=os.environ["DATABRICKS_WAREHOUSE_ID"],
28
28
  token=os.environ["DATABRICKS_TOKEN"],
29
29
  )
30
- result = await client.execute_arrow("SELECT * FROM my_catalog.my_schema.my_table LIMIT 1000000") # noqa: F841 -- read by DuckDB's replacement scan via local variable name, not a literal Python reference
30
+ result_set = await client.execute("SELECT * FROM my_catalog.my_schema.my_table LIMIT 1000000")
31
+ result = await result_set.fetchall_arrow() # noqa: F841 -- read by DuckDB's replacement scan via local variable name, not a literal Python reference
31
32
 
32
33
  # `result` is queryable by variable name -- DuckDB imports it zero-copy,
33
34
  # no pandas/pyarrow conversion step in between.
@@ -56,6 +56,15 @@ pub struct ColumnDescription {
56
56
  pub name: String,
57
57
  #[serde(default)]
58
58
  pub type_name: Option<String>,
59
+ /// Only present for `type_name == "DECIMAL"` -- needed by
60
+ /// `json_convert`'s INLINE/JSON_ARRAY-to-Arrow conversion to build a
61
+ /// correctly-scaled `Decimal128Array`. Absent (and unused) for every
62
+ /// other type, including on the Arrow-IPC path, which gets its decimal
63
+ /// precision/scale from the IPC schema itself, not from here.
64
+ #[serde(default)]
65
+ pub type_precision: Option<u8>,
66
+ #[serde(default)]
67
+ pub type_scale: Option<i8>,
59
68
  }
60
69
 
61
70
  #[derive(Deserialize, Default)]
@@ -119,6 +128,14 @@ struct ResultLinkBody {
119
128
  struct ResultBody {
120
129
  #[serde(default)]
121
130
  external_links: Vec<ResultLinkBody>,
131
+ /// Only present for `disposition: INLINE` + `format: JSON_ARRAY` -- each
132
+ /// row a `Vec<Option<String>>` (every non-null value a string,
133
+ /// Databricks' own JSON_ARRAY contract, same as `execute_json_statement`'s
134
+ /// normal `EXTERNAL_LINKS`+`JSON_ARRAY` chunks). Consumed by
135
+ /// `json_convert::json_array_to_record_batch` on the `prefer_inline`
136
+ /// fast path.
137
+ #[serde(default)]
138
+ data_array: Option<Vec<Vec<Option<String>>>>,
122
139
  }
123
140
 
124
141
  #[derive(Deserialize)]
@@ -197,7 +214,18 @@ impl ApiError {
197
214
  /// (excluding `is_connect()`/`is_timeout()`, already handled) also covers
198
215
  /// hyper's own "connection closed before message completed" pooled-
199
216
  /// connection-reuse race, which fires before any body is even read and
200
- /// is equally safe to retry on an idempotent request.
217
+ /// is equally safe to retry on an idempotent request. Unlike `is_decode()`
218
+ /// (hit for real on a 400-chunk fetch and reproduced with a directed
219
+ /// test), this specific race is reasoned from reqwest/hyper's own source,
220
+ /// not independently reproduced -- a directed attempt to force it
221
+ /// self-healed 20/20 times (hyper's own idle-connection health check
222
+ /// evidently detects a closed peer and opens a fresh connection before
223
+ /// ever handing the dead one out for reuse, at least under a simple,
224
+ /// low-concurrency test). Kept anyway since it's safe regardless (scoped
225
+ /// to idempotent requests only) and the window may still be reachable
226
+ /// under real production concurrency even though a quick local test
227
+ /// couldn't force it -- treat as a plausible defensive measure, not a
228
+ /// confirmed fix for an observed failure.
201
229
  fn from_reqwest(e: reqwest::Error, idempotent: bool) -> Self {
202
230
  // `is_decode()` (reqwest's `Kind::Decode`) isn't only content-decoding
203
231
  // -- `Response::bytes()`/`.text()`/`.json()` also wrap a body read
@@ -219,8 +247,18 @@ impl ApiError {
219
247
  }
220
248
  }
221
249
 
222
- fn from_status(status: StatusCode, body: &str) -> Self {
223
- let transient = matches!(status.as_u16(), 401 | 403 | 408 | 429) || status.is_server_error();
250
+ /// `idempotent` only gates the 5xx case -- 401/403/408/429 mean the
251
+ /// request was rejected before any processing started (auth failure,
252
+ /// rate limit, client-side timeout), safe to retry regardless of method.
253
+ /// A 5xx is murkier: it usually means the same, but it can also mean the
254
+ /// backend already accepted and started the statement before some later
255
+ /// failure (e.g. a gateway timeout) produced the 5xx anyway -- found in
256
+ /// code review that this call was unconditionally transient even for the
257
+ /// statement-submit POST, bypassing the exact idempotency reasoning
258
+ /// `from_reqwest`'s `idempotent` param exists for (retrying that POST
259
+ /// risks a second, duplicate execution of arbitrary caller SQL).
260
+ fn from_status(status: StatusCode, body: &str, idempotent: bool) -> Self {
261
+ let transient = matches!(status.as_u16(), 401 | 403 | 408 | 429) || (idempotent && status.is_server_error());
224
262
  Self {
225
263
  message: format!("HTTP {status}: {body}"),
226
264
  transient,
@@ -271,6 +309,23 @@ pub struct StatementSubmitResult {
271
309
  pub compressed: bool,
272
310
  }
273
311
 
312
+ /// What `execute_arrow_statement_prefer_inline` gets you -- either the whole
313
+ /// result inline as raw JSON_ARRAY rows (every non-null value a string,
314
+ /// same contract as `execute_json_statement`; converting these into a real
315
+ /// `RecordBatch` is `pipeline.rs`'s job via `json_convert`, not this
316
+ /// Arrow-agnostic module's -- see this file's own module doc comment), or a
317
+ /// normal `StatementSubmitResult` to fetch chunks for exactly as if
318
+ /// `prefer_inline` had never been asked for. See that function's own doc
319
+ /// comment for when each happens.
320
+ pub enum InlineOrExternal {
321
+ Inline {
322
+ statement_id: String,
323
+ rows: Vec<Vec<Option<String>>>,
324
+ columns: Vec<ColumnDescription>,
325
+ },
326
+ External(StatementSubmitResult),
327
+ }
328
+
274
329
  #[derive(Debug)]
275
330
  pub struct ChunkItem {
276
331
  /// `Bytes` (not `Vec<u8>`) so the bytes reqwest already received off the
@@ -316,14 +371,22 @@ fn decompress_lz4_frame(compressed: &Bytes) -> Result<Bytes, ApiError> {
316
371
  // small lower bound.
317
372
  let mut out = Vec::with_capacity(compressed.len() * 4);
318
373
  let mut decoder = lz4_flex::frame::FrameDecoder::new(&compressed[..]);
319
- loop {
320
- let before = out.len();
374
+ // Terminate on the *reader* being exhausted, not on "output stopped
375
+ // growing" -- found in code review that a frame which happens to decode
376
+ // to zero bytes (a real, valid LZ4 Frame shape: header + immediate
377
+ // EndMark) makes one `read_to_end` call return `Ok(0)` for that frame
378
+ // without erroring and without necessarily advancing into the next
379
+ // frame yet, which the old `out.len() == before` check read as "no more
380
+ // frames" -- silently dropping every subsequent concatenated frame with
381
+ // no error at all. Same silent-truncation shape as the original
382
+ // multi-frame bug this loop exists to fix. Looping while the reader
383
+ // still has bytes left (regardless of whether the last call grew `out`)
384
+ // is the correct fix -- verified against a zero-content frame sandwiched
385
+ // between two real ones, see `decompress_lz4_frame_survives_a_zero_content_frame_in_the_middle`.
386
+ while !decoder.get_ref().is_empty() {
321
387
  decoder
322
388
  .read_to_end(&mut out)
323
389
  .map_err(|e| ApiError::permanent(format!("LZ4 frame decompress failed: {e}")))?;
324
- if out.len() == before {
325
- break;
326
- }
327
390
  }
328
391
  Ok(Bytes::from(out))
329
392
  }
@@ -499,7 +562,7 @@ impl DbClient {
499
562
  let status = resp.status();
500
563
  let text = resp.text().await.map_err(|e| ApiError::from_reqwest(e, idempotent))?;
501
564
  if !status.is_success() {
502
- return Err(ApiError::from_status(status, &text));
565
+ return Err(ApiError::from_status(status, &text, idempotent));
503
566
  }
504
567
  serde_json::from_str::<T>(&text).map_err(|e| ApiError::permanent(format!("bad JSON body: {e}")))
505
568
  })
@@ -530,7 +593,7 @@ impl DbClient {
530
593
  let status = resp.status();
531
594
  if !status.is_success() {
532
595
  let text = resp.text().await.unwrap_or_default();
533
- return Err(ApiError::from_status(status, &text));
596
+ return Err(ApiError::from_status(status, &text, true));
534
597
  }
535
598
  let bytes = resp.bytes().await.map_err(|e| ApiError::from_reqwest(e, true))?;
536
599
  if !compressed {
@@ -595,6 +658,101 @@ impl DbClient {
595
658
  .await
596
659
  }
597
660
 
661
+ /// Tries `disposition: INLINE` + `format: JSON_ARRAY` first -- for a
662
+ /// small result, Databricks embeds the whole result directly in this
663
+ /// same submit/poll response (`result.data_array`), skipping the
664
+ /// separate chunk-resolution-and-blob-fetch round trip the normal
665
+ /// `EXTERNAL_LINKS` path always needs. Confirmed against a real
666
+ /// workspace: exceeding INLINE's byte limit (26,214,400 bytes / 25MiB)
667
+ /// fails the statement cleanly with a specific, matchable error
668
+ /// message -- never silent truncation -- and `INLINE_OR_EXTERNAL_LINKS`
669
+ /// ("HYBRID", which would let the server choose per-query) returned a
670
+ /// clean "not a supported disposition" 400 on that same workspace, so
671
+ /// isn't used here. `format` must be `JSON_ARRAY` for `INLINE` --
672
+ /// Databricks rejects `INLINE`+`ARROW_STREAM` outright (also confirmed).
673
+ /// This module stays Arrow-agnostic on purpose (see its own module doc
674
+ /// comment) -- `InlineOrExternal::Inline` carries the raw JSON_ARRAY
675
+ /// rows straight through; converting them into a `RecordBatch` (and
676
+ /// falling back to a fresh `EXTERNAL_LINKS` submission if that
677
+ /// conversion hits a column type it doesn't handle) is `pipeline.rs`'s
678
+ /// job via `json_convert`, same division of responsibility as every
679
+ /// other decode step in this crate.
680
+ ///
681
+ /// Falls back to a **fresh, independent** `execute_arrow_statement` call
682
+ /// (not a retry of this same submission) whenever INLINE doesn't pan
683
+ /// out at the HTTP/statement level: the byte limit was exceeded, or
684
+ /// `data_array` is unexpectedly absent despite SUCCEEDED. This is a
685
+ /// second, distinct statement execution, not a retry of a possibly-
686
+ /// already-run one, so it carries none of the double-execution risk
687
+ /// that made POST retries unsafe elsewhere in this file -- the first
688
+ /// attempt's outcome is fully known (it reached a terminal state)
689
+ /// before the second one is ever submitted. A genuine query error (bad
690
+ /// SQL, permission denied) propagates immediately instead, same as the
691
+ /// normal path -- no reason to mask it behind a pointless second
692
+ /// attempt.
693
+ ///
694
+ /// Not the default -- opt-in only (`prefer_inline` on the Python-facing
695
+ /// `execute()`/`Cursor.execute()`), since a caller who doesn't expect a
696
+ /// small result pays for two full statement executions on the (common,
697
+ /// for them) fallback path instead of one.
698
+ pub async fn execute_arrow_statement_prefer_inline(
699
+ &self,
700
+ statement: &str,
701
+ catalog: Option<&str>,
702
+ schema: Option<&str>,
703
+ parameters: Option<Value>,
704
+ ) -> Result<InlineOrExternal, ApiError> {
705
+ let mut body = json!({
706
+ "warehouse_id": self.warehouse_id,
707
+ "statement": statement,
708
+ "disposition": "INLINE",
709
+ "format": "JSON_ARRAY",
710
+ "wait_timeout": self.wait_timeout,
711
+ "on_wait_timeout": "CONTINUE",
712
+ });
713
+ if let Some(c) = catalog {
714
+ body["catalog"] = json!(c);
715
+ }
716
+ if let Some(s) = schema {
717
+ body["schema"] = json!(s);
718
+ }
719
+ if let Some(p) = parameters.clone() {
720
+ body["parameters"] = p;
721
+ }
722
+
723
+ let outcome = self.submit_and_poll(body).await;
724
+ let data = match outcome {
725
+ Ok(d) => d,
726
+ Err(e) if e.message.contains("Inline byte limit exceeded") => {
727
+ return self
728
+ .execute_arrow_statement(statement, catalog, schema, parameters)
729
+ .await
730
+ .map(InlineOrExternal::External);
731
+ }
732
+ Err(e) => return Err(e),
733
+ };
734
+
735
+ let manifest = data.manifest.unwrap_or_default();
736
+ let columns = manifest.schema.map(|s| s.columns).unwrap_or_default();
737
+ let data_array = data.result.and_then(|r| r.data_array);
738
+ match data_array {
739
+ Some(rows) => Ok(InlineOrExternal::Inline {
740
+ statement_id: data.statement_id,
741
+ rows,
742
+ columns,
743
+ }),
744
+ // No data_array despite SUCCEEDED -- shouldn't happen given a
745
+ // non-error status, but this crate never guesses at a missing
746
+ // field; a fresh EXTERNAL_LINKS submission is exactly as safe as
747
+ // the byte-limit fallback above (a distinct statement, not a
748
+ // retry of this one).
749
+ None => self
750
+ .execute_arrow_statement(statement, catalog, schema, parameters)
751
+ .await
752
+ .map(InlineOrExternal::External),
753
+ }
754
+ }
755
+
598
756
  /// Like `execute_arrow_statement`, fixed to JSON_ARRAY -- each fetched
599
757
  /// chunk's bytes are then a JSON array of rows, each row itself an array
600
758
  /// of values where every non-null value is a *string* regardless of its
@@ -613,6 +771,41 @@ impl DbClient {
613
771
  .await
614
772
  }
615
773
 
774
+ /// Shared by `execute_statement` and `execute_arrow_statement_prefer_inline`:
775
+ /// POST the statement, poll until a terminal state, and turn FAILED/
776
+ /// CANCELED into an `Err` -- everything both callers need before they
777
+ /// diverge on how to interpret a SUCCEEDED response's `result`/`manifest`.
778
+ async fn submit_and_poll(&self, body: Value) -> Result<StatementResponseBody, ApiError> {
779
+ self.ensure_warehouse_running().await?;
780
+ let url = format!("{}/api/2.0/sql/statements", self.host);
781
+ let mut data: StatementResponseBody = self.authed_json(reqwest::Method::POST, &url, Some(&body)).await?;
782
+
783
+ while !matches!(
784
+ data.status.state.as_str(),
785
+ "SUCCEEDED" | "FAILED" | "CANCELED" | "CLOSED"
786
+ ) {
787
+ tokio::time::sleep(POLL_INTERVAL).await;
788
+ let poll_url = format!("{}/api/2.0/sql/statements/{}", self.host, data.statement_id);
789
+ data = self.authed_json(reqwest::Method::GET, &poll_url, None).await?;
790
+ }
791
+
792
+ match data.status.state.as_str() {
793
+ "FAILED" => {
794
+ let err = data.status.error.unwrap_or(StatementErrorBody {
795
+ error_code: None,
796
+ message: None,
797
+ });
798
+ Err(ApiError::permanent(format!(
799
+ "Databricks statement failed [{}]: {}",
800
+ err.error_code.as_deref().unwrap_or(""),
801
+ err.message.as_deref().unwrap_or(""),
802
+ )))
803
+ }
804
+ "CANCELED" => Err(ApiError::permanent("Databricks statement was canceled")),
805
+ _ => Ok(data),
806
+ }
807
+ }
808
+
616
809
  async fn execute_statement(
617
810
  &self,
618
811
  statement: &str,
@@ -649,35 +842,7 @@ impl DbClient {
649
842
  body["parameters"] = p;
650
843
  }
651
844
 
652
- self.ensure_warehouse_running().await?;
653
- let url = format!("{}/api/2.0/sql/statements", self.host);
654
- let mut data: StatementResponseBody = self.authed_json(reqwest::Method::POST, &url, Some(&body)).await?;
655
-
656
- while !matches!(
657
- data.status.state.as_str(),
658
- "SUCCEEDED" | "FAILED" | "CANCELED" | "CLOSED"
659
- ) {
660
- tokio::time::sleep(POLL_INTERVAL).await;
661
- let poll_url = format!("{}/api/2.0/sql/statements/{}", self.host, data.statement_id);
662
- data = self.authed_json(reqwest::Method::GET, &poll_url, None).await?;
663
- }
664
-
665
- match data.status.state.as_str() {
666
- "FAILED" => {
667
- let err = data.status.error.unwrap_or(StatementErrorBody {
668
- error_code: None,
669
- message: None,
670
- });
671
- return Err(ApiError::permanent(format!(
672
- "Databricks statement failed [{}]: {}",
673
- err.error_code.as_deref().unwrap_or(""),
674
- err.message.as_deref().unwrap_or(""),
675
- )));
676
- }
677
- "CANCELED" => return Err(ApiError::permanent("Databricks statement was canceled")),
678
- _ => {}
679
- }
680
-
845
+ let data = self.submit_and_poll(body).await?;
681
846
  let manifest = data.manifest.unwrap_or_default();
682
847
  let compressed = manifest.result_compression.as_deref() == Some("LZ4_FRAME");
683
848
  // `Vec` per index, not a plain map entry -- see `ChunkMeta::pre_resolved_links`'s
@@ -837,7 +1002,7 @@ impl DbClient {
837
1002
  let status = resp.status();
838
1003
  if !status.is_success() {
839
1004
  let text = resp.text().await.unwrap_or_default();
840
- return Err(ApiError::from_status(status, &text));
1005
+ return Err(ApiError::from_status(status, &text, true));
841
1006
  }
842
1007
  Ok(())
843
1008
  })
@@ -866,7 +1031,7 @@ impl DbClient {
866
1031
  return Ok(());
867
1032
  }
868
1033
  let text = resp.text().await.unwrap_or_default();
869
- Err(ApiError::from_status(status, &text))
1034
+ Err(ApiError::from_status(status, &text, true))
870
1035
  })
871
1036
  .await
872
1037
  }
@@ -932,6 +1097,44 @@ mod tests {
932
1097
  assert_eq!(decompressed, Bytes::from(expected));
933
1098
  }
934
1099
 
1100
+ /// Regression test for a bug found in code review: a real, valid LZ4
1101
+ /// Frame that happens to decode to zero bytes (a header immediately
1102
+ /// followed by an EndMark -- a legal frame shape, not malformed input)
1103
+ /// makes `read_to_end` return `Ok(0)` for that frame without erroring.
1104
+ /// The old loop read "output didn't grow" as "no more frames" and
1105
+ /// stopped there, silently dropping every frame concatenated after the
1106
+ /// empty one. `decompress_lz4_frame` must keep going as long as the
1107
+ /// underlying reader still has bytes left, not just as long as output
1108
+ /// keeps growing.
1109
+ #[test]
1110
+ fn decompress_lz4_frame_survives_a_zero_content_frame_in_the_middle() {
1111
+ use std::io::Write;
1112
+
1113
+ fn compress_one_frame(data: &[u8]) -> Vec<u8> {
1114
+ let mut encoder = lz4_flex::frame::FrameEncoder::new(Vec::new());
1115
+ encoder.write_all(data).unwrap();
1116
+ encoder.finish().unwrap()
1117
+ }
1118
+
1119
+ let part_a = b"the quick brown fox jumps over the lazy dog ".repeat(50);
1120
+ let part_b = b"pack my box with five dozen liquor jugs ".repeat(50);
1121
+ let mut concatenated_frames = Vec::new();
1122
+ concatenated_frames.extend(compress_one_frame(&part_a));
1123
+ concatenated_frames.extend(compress_one_frame(b"")); // real frame, zero content
1124
+ concatenated_frames.extend(compress_one_frame(&part_b));
1125
+
1126
+ let decompressed = decompress_lz4_frame(&Bytes::from(concatenated_frames)).unwrap();
1127
+
1128
+ let mut expected = Vec::new();
1129
+ expected.extend_from_slice(&part_a);
1130
+ expected.extend_from_slice(&part_b);
1131
+ assert_eq!(
1132
+ decompressed,
1133
+ Bytes::from(expected),
1134
+ "the frame after the zero-content one must not be silently dropped"
1135
+ );
1136
+ }
1137
+
935
1138
  fn ok_task() -> tokio::task::JoinHandle<Result<(), ApiError>> {
936
1139
  tokio::spawn(async { Ok(()) })
937
1140
  }