arrowbricks 1.3.2__tar.gz → 1.4.0__tar.gz
This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
- {arrowbricks-1.3.2 → arrowbricks-1.4.0}/PKG-INFO +3 -3
- {arrowbricks-1.3.2 → arrowbricks-1.4.0}/README.md +2 -2
- {arrowbricks-1.3.2 → arrowbricks-1.4.0}/pyproject.toml +3 -3
- {arrowbricks-1.3.2 → arrowbricks-1.4.0}/rust/arrowbricks_core/Cargo.lock +2 -1
- {arrowbricks-1.3.2 → arrowbricks-1.4.0}/rust/arrowbricks_core/Cargo.toml +6 -1
- {arrowbricks-1.3.2 → arrowbricks-1.4.0}/rust/arrowbricks_core/README.md +6 -7
- {arrowbricks-1.3.2 → arrowbricks-1.4.0}/rust/arrowbricks_core/examples/duckdb_query.py +2 -1
- {arrowbricks-1.3.2 → arrowbricks-1.4.0}/rust/arrowbricks_core/src/client.rs +244 -41
- arrowbricks-1.4.0/rust/arrowbricks_core/src/json_convert.rs +476 -0
- {arrowbricks-1.3.2 → arrowbricks-1.4.0}/rust/arrowbricks_core/src/lib.rs +66 -154
- {arrowbricks-1.3.2 → arrowbricks-1.4.0}/rust/arrowbricks_core/src/pipeline.rs +178 -3
- {arrowbricks-1.3.2 → arrowbricks-1.4.0}/rust/arrowbricks_core/tests/wiremock_pipeline.rs +354 -1
- {arrowbricks-1.3.2 → arrowbricks-1.4.0}/rust/arrowbricks_core/tests_py/test_parameters.py +3 -16
- {arrowbricks-1.3.2 → arrowbricks-1.4.0}/rust/arrowbricks_core/tests_py/test_streaming.py +42 -33
- {arrowbricks-1.3.2 → arrowbricks-1.4.0}/rust/arrowbricks_core/tests_py/test_token_provider.py +52 -2
- {arrowbricks-1.3.2 → arrowbricks-1.4.0}/src/arrowbricks/_core.pyi +1 -27
- {arrowbricks-1.3.2 → arrowbricks-1.4.0}/src/arrowbricks/client.py +18 -18
- {arrowbricks-1.3.2 → arrowbricks-1.4.0}/src/arrowbricks/cursor.py +33 -3
- arrowbricks-1.3.2/rust/arrowbricks_core/tests_py/test_execute_json.py +0 -101
- {arrowbricks-1.3.2 → arrowbricks-1.4.0}/LICENSE +0 -0
- {arrowbricks-1.3.2 → arrowbricks-1.4.0}/rust/arrowbricks_core/.gitignore +0 -0
- {arrowbricks-1.3.2 → arrowbricks-1.4.0}/rust/arrowbricks_core/examples/fastapi_sse.py +0 -0
- {arrowbricks-1.3.2 → arrowbricks-1.4.0}/rust/arrowbricks_core/rustfmt.toml +0 -0
- {arrowbricks-1.3.2 → arrowbricks-1.4.0}/rust/arrowbricks_core/src/heartbeat.rs +0 -0
- {arrowbricks-1.3.2 → arrowbricks-1.4.0}/rust/arrowbricks_core/tests/wiremock_volume_files.rs +0 -0
- {arrowbricks-1.3.2 → arrowbricks-1.4.0}/rust/arrowbricks_core/tests_py/test_ipc_stream.py +0 -0
- {arrowbricks-1.3.2 → arrowbricks-1.4.0}/rust/arrowbricks_core/tests_py/test_stream_ndjson_lines.py +0 -0
- {arrowbricks-1.3.2 → arrowbricks-1.4.0}/rust/arrowbricks_core/tests_py/test_volume_files.py +0 -0
- {arrowbricks-1.3.2 → arrowbricks-1.4.0}/src/arrowbricks/__init__.py +0 -0
- {arrowbricks-1.3.2 → arrowbricks-1.4.0}/src/arrowbricks/_streaming.py +0 -0
- {arrowbricks-1.3.2 → arrowbricks-1.4.0}/src/arrowbricks/py.typed +0 -0
|
@@ -1,6 +1,6 @@
|
|
|
1
1
|
Metadata-Version: 2.4
|
|
2
2
|
Name: arrowbricks
|
|
3
|
-
Version: 1.
|
|
3
|
+
Version: 1.4.0
|
|
4
4
|
Requires-Dist: arro3-core>=0.8 ; extra == 'arro3'
|
|
5
5
|
Provides-Extra: arro3
|
|
6
6
|
License-File: LICENSE
|
|
@@ -139,7 +139,7 @@ conn = connect(host=..., warehouse_id=..., token_provider=my_token_provider)
|
|
|
139
139
|
- `connect(host, warehouse_id, *, token=None, token_provider=None, ...) -> Connection`
|
|
140
140
|
- `Connection.cursor() -> Cursor`
|
|
141
141
|
- `Connection.client -> DatabricksClient` -- the same client `cursor()` uses, for lower-level access (e.g. `stream_query_json`, `upload_volume_file`).
|
|
142
|
-
- `Cursor.execute(sql, parameters=None, *, row_limit=None, offset=None, catalog=None, schema=None, total_timeout_s=None) -> Cursor` -- submits and waits for the statement, like a real DB-API cursor. `parameters`, if given, is Databricks' own named-parameter format -- `[{"name": ..., "value": ..., "type": ...}]` bound against `:name` markers in `sql`.
|
|
142
|
+
- `Cursor.execute(sql, parameters=None, *, row_limit=None, offset=None, catalog=None, schema=None, total_timeout_s=None, prefer_inline=False) -> Cursor` -- submits and waits for the statement, like a real DB-API cursor. `parameters`, if given, is Databricks' own named-parameter format -- `[{"name": ..., "value": ..., "type": ...}]` bound against `:name` markers in `sql`. `prefer_inline=True` tries fetching a small result (well under Databricks' 25 MiB inline cap) in the same round trip as the submission itself, skipping the chunk-fetch entirely -- if the result turns out too big, or has a column type this can't convert (nested ARRAY/MAP/STRUCT, VARIANT), it transparently re-runs the query the normal way, so a caller who sets this without actually expecting a small result pays for the query twice. Leave it off unless you know the result is small.
|
|
143
143
|
- `Cursor.execute_streamed(...)` -- same args, but an async generator yielding `HEARTBEAT` while waiting on a slow cold start, then the ready `Cursor` -- for bridging e.g. an SSE connection. Its timeout/heartbeats stop the moment the statement is ready, *before* any chunk has been downloaded -- see `fetchall_streamed` below for the download phase itself.
|
|
144
144
|
- `Cursor.fetchone() -> tuple | None`, `Cursor.fetchmany(size) -> list[tuple]`, `Cursor.fetchall() -> list[tuple]`, and iterating a `Cursor` directly -- row tuples; needs the `arro3` extra.
|
|
145
145
|
- `Cursor.fetchmany_arrow(size) -> Table`, `Cursor.fetchall_arrow() -> Table` -- an Arrow table (implements `__arrow_c_stream__`, so arro3/pyarrow/DuckDB can all consume it directly, zero-copy).
|
|
@@ -191,7 +191,7 @@ con.register("my_table", table)
|
|
|
191
191
|
con.sql("SELECT count(*) FROM my_table").show()
|
|
192
192
|
```
|
|
193
193
|
|
|
194
|
-
`ReplayableArrowChunk` works the same way for Arrow-IPC bytes you fetched and stored earlier (e.g.
|
|
194
|
+
`ReplayableArrowChunk` works the same way for Arrow-IPC bytes you fetched and stored earlier (e.g. a raw chunk's bytes, cached in Redis/a file/wherever) -- DuckDB's registration path calls `__arrow_c_stream__` twice (a schema peek, then the actual scan), which is exactly what `ReplayableArrowChunk` exists to support:
|
|
195
195
|
|
|
196
196
|
```python
|
|
197
197
|
from arrowbricks import ReplayableArrowChunk
|
|
@@ -127,7 +127,7 @@ conn = connect(host=..., warehouse_id=..., token_provider=my_token_provider)
|
|
|
127
127
|
- `connect(host, warehouse_id, *, token=None, token_provider=None, ...) -> Connection`
|
|
128
128
|
- `Connection.cursor() -> Cursor`
|
|
129
129
|
- `Connection.client -> DatabricksClient` -- the same client `cursor()` uses, for lower-level access (e.g. `stream_query_json`, `upload_volume_file`).
|
|
130
|
-
- `Cursor.execute(sql, parameters=None, *, row_limit=None, offset=None, catalog=None, schema=None, total_timeout_s=None) -> Cursor` -- submits and waits for the statement, like a real DB-API cursor. `parameters`, if given, is Databricks' own named-parameter format -- `[{"name": ..., "value": ..., "type": ...}]` bound against `:name` markers in `sql`.
|
|
130
|
+
- `Cursor.execute(sql, parameters=None, *, row_limit=None, offset=None, catalog=None, schema=None, total_timeout_s=None, prefer_inline=False) -> Cursor` -- submits and waits for the statement, like a real DB-API cursor. `parameters`, if given, is Databricks' own named-parameter format -- `[{"name": ..., "value": ..., "type": ...}]` bound against `:name` markers in `sql`. `prefer_inline=True` tries fetching a small result (well under Databricks' 25 MiB inline cap) in the same round trip as the submission itself, skipping the chunk-fetch entirely -- if the result turns out too big, or has a column type this can't convert (nested ARRAY/MAP/STRUCT, VARIANT), it transparently re-runs the query the normal way, so a caller who sets this without actually expecting a small result pays for the query twice. Leave it off unless you know the result is small.
|
|
131
131
|
- `Cursor.execute_streamed(...)` -- same args, but an async generator yielding `HEARTBEAT` while waiting on a slow cold start, then the ready `Cursor` -- for bridging e.g. an SSE connection. Its timeout/heartbeats stop the moment the statement is ready, *before* any chunk has been downloaded -- see `fetchall_streamed` below for the download phase itself.
|
|
132
132
|
- `Cursor.fetchone() -> tuple | None`, `Cursor.fetchmany(size) -> list[tuple]`, `Cursor.fetchall() -> list[tuple]`, and iterating a `Cursor` directly -- row tuples; needs the `arro3` extra.
|
|
133
133
|
- `Cursor.fetchmany_arrow(size) -> Table`, `Cursor.fetchall_arrow() -> Table` -- an Arrow table (implements `__arrow_c_stream__`, so arro3/pyarrow/DuckDB can all consume it directly, zero-copy).
|
|
@@ -179,7 +179,7 @@ con.register("my_table", table)
|
|
|
179
179
|
con.sql("SELECT count(*) FROM my_table").show()
|
|
180
180
|
```
|
|
181
181
|
|
|
182
|
-
`ReplayableArrowChunk` works the same way for Arrow-IPC bytes you fetched and stored earlier (e.g.
|
|
182
|
+
`ReplayableArrowChunk` works the same way for Arrow-IPC bytes you fetched and stored earlier (e.g. a raw chunk's bytes, cached in Redis/a file/wherever) -- DuckDB's registration path calls `__arrow_c_stream__` twice (a schema peek, then the actual scan), which is exactly what `ReplayableArrowChunk` exists to support:
|
|
183
183
|
|
|
184
184
|
```python
|
|
185
185
|
from arrowbricks import ReplayableArrowChunk
|
|
@@ -1,6 +1,6 @@
|
|
|
1
1
|
[project]
|
|
2
2
|
name = "arrowbricks"
|
|
3
|
-
version = "1.
|
|
3
|
+
version = "1.4.0"
|
|
4
4
|
description = "Runs SQL against a Databricks SQL warehouse via the Statement Execution API and hands you the result as Arrow -- a DB-API-ish Cursor (fetchone/fetchmany/fetchall/fetchall_arrow) or NDJSON streaming. Rust/PyO3 core throughout -- zero required runtime dependencies."
|
|
5
5
|
readme = "README.md"
|
|
6
6
|
license = "MIT"
|
|
@@ -10,8 +10,8 @@ requires-python = ">=3.11"
|
|
|
10
10
|
# (see rust/arrowbricks_core's own reqwest client + Arrow-IPC/NDJSON
|
|
11
11
|
# writer). arro3-core is optional (see below): only Cursor.fetchone/
|
|
12
12
|
# fetchmany/fetchall's row materialization needs it -- fetchall_arrow,
|
|
13
|
-
#
|
|
14
|
-
#
|
|
13
|
+
# stream_query_json, upload_volume_file/delete_volume_file, and
|
|
14
|
+
# Cursor.description all work with zero dependencies installed.
|
|
15
15
|
dependencies = []
|
|
16
16
|
|
|
17
17
|
[project.optional-dependencies]
|
|
@@ -240,11 +240,12 @@ dependencies = [
|
|
|
240
240
|
|
|
241
241
|
[[package]]
|
|
242
242
|
name = "arrowbricks_core"
|
|
243
|
-
version = "1.
|
|
243
|
+
version = "1.4.0"
|
|
244
244
|
dependencies = [
|
|
245
245
|
"arrow",
|
|
246
246
|
"arrow-json",
|
|
247
247
|
"bytes",
|
|
248
|
+
"chrono",
|
|
248
249
|
"lz4_flex",
|
|
249
250
|
"pyo3",
|
|
250
251
|
"pyo3-arrow",
|
|
@@ -1,6 +1,6 @@
|
|
|
1
1
|
[package]
|
|
2
2
|
name = "arrowbricks_core"
|
|
3
|
-
version = "1.
|
|
3
|
+
version = "1.4.0"
|
|
4
4
|
edition = "2024"
|
|
5
5
|
readme = "README.md"
|
|
6
6
|
|
|
@@ -15,6 +15,11 @@ crate-type = ["cdylib", "rlib"]
|
|
|
15
15
|
arrow = { version = "59.1.0", default-features = false, features = ["ipc"] }
|
|
16
16
|
arrow-json = "59.1.0"
|
|
17
17
|
bytes = "1.12.1"
|
|
18
|
+
# Already an unconditional transitive dependency of pyo3-arrow (see its own
|
|
19
|
+
# comment below) -- declared directly here too so json_convert.rs can parse
|
|
20
|
+
# DATE/TIMESTAMP/TIMESTAMP_NTZ strings without hand-rolling calendar
|
|
21
|
+
# arithmetic. No new code ships that wasn't already linked in.
|
|
22
|
+
chrono = { version = "0.4", default-features = false, features = ["std"] }
|
|
18
23
|
# Decompresses cloud-fetch chunk bytes when the server honors our
|
|
19
24
|
# `result_compression: LZ4_FRAME` request (see client.rs's execute_statement)
|
|
20
25
|
# -- the same LZ4 Frame format databricks-sql-python's `lz4.frame` decodes.
|
|
@@ -48,7 +48,8 @@ async def main():
|
|
|
48
48
|
warehouse_id="abcd1234efgh5678",
|
|
49
49
|
token="dapi...",
|
|
50
50
|
)
|
|
51
|
-
|
|
51
|
+
result = await client.execute("SELECT * FROM my_catalog.my_schema.my_table LIMIT 100")
|
|
52
|
+
table = await result.fetchall_arrow()
|
|
52
53
|
print(table.num_rows)
|
|
53
54
|
|
|
54
55
|
|
|
@@ -66,17 +67,14 @@ unlike the table object itself.
|
|
|
66
67
|
## API
|
|
67
68
|
|
|
68
69
|
- `Client(host, warehouse_id, *, token=None, token_provider=None, chunk_fetch_concurrency=64, http_timeout=60.0, wait_timeout="30s", warehouse_start_timeout=300.0, warehouse_confirmed_running_ttl_s=30.0, compress_results=True)` -- exactly one of `token`/`token_provider`. `token_provider` is a callable (sync or async) returning a token string, called fresh on every request, no caching. `compress_results` requests LZ4-compressed cloud-fetch chunks (see above); set `False` to opt out.
|
|
69
|
-
- `Client.
|
|
70
|
-
- `Client.execute(statement, *, catalog=None, schema=None, parameters=None) -> ResultSet` -- submits and starts background chunk fetching without pulling anything yet.
|
|
70
|
+
- `Client.execute(statement, *, catalog=None, schema=None, parameters=None, prefer_inline=False) -> ResultSet` -- submits and starts background chunk fetching without pulling anything yet. `parameters` is Databricks' own named-parameter format (`[{"name":..., "value":..., "type":...}]`), passed straight through. `prefer_inline=True` submits with `disposition=INLINE, format=JSON_ARRAY` instead, for a caller who expects a small (well under Databricks' 25 MiB inline cap) result and wants to skip the chunk-fetch round trip -- on a result too big for INLINE, or containing a column type `json_convert.rs` doesn't map (STRUCT/ARRAY-of-STRUCT/MAP/VARIANT), it transparently falls back to a second, normal `execute()` and the query runs twice. See `AGENTS.md`'s "Design invariants" section (in the root package) for the full reasoning and real-workspace verification behind this.
|
|
71
71
|
- `ResultSet.fetchmany_arrow(n) -> Table` -- pulls/decodes only as many chunks as needed for `n` rows, buffering the rest; may return fewer than `n` once exhausted.
|
|
72
72
|
- `ResultSet.fetchall_arrow() -> Table` -- drains everything remaining.
|
|
73
|
+
- `ResultSet.fetchall_arrow_streamed(*, total_timeout_s=None)` -- same as `fetchall_arrow()`, but an async iterator yielding the `HEARTBEAT` singleton while pulling chunks instead of blocking silently (bridge e.g. an SSE connection through the download), then a `Table` exactly once. Raises if `total_timeout_s` elapses first.
|
|
73
74
|
- `ResultSet.schema() -> list[tuple[str, str]] | None` -- the real decoded Arrow schema as `(name, type_name)` string pairs, once known (after >=1 fetch); `None` before that. Computed directly in Rust -- unlike a `Table`'s own `.schema`, this needs no arro3 install.
|
|
74
75
|
- `ResultSet.statement_id`, `ResultSet.num_chunks`, `ResultSet.columns` (manifest-based pre-fetch schema estimate, also arro3-free).
|
|
75
|
-
- `Client.execute_json(statement, *, catalog=None, schema=None, parameters=None) -> list[list[str | None]]` -- JSON_ARRAY format, no Arrow parse. Every non-null value comes back as a string regardless of real column type (Databricks' own contract) -- cast by the manifest's column type yourself if you want native Python types.
|
|
76
76
|
- `Client.stream_ndjson_lines(statement, *, catalog=None, schema=None, parameters=None, total_timeout_s=None)` -- chunk-at-a-time: an async iterator yielding the `HEARTBEAT` singleton while waiting on the statement or any individual chunk, then a `list[str]` of NDJSON lines (one per row, explicit nulls, ISO-8601 timestamps) per chunk in logical order. Decode and JSON encoding both happen in Rust -- backs `stream_query_json` end to end.
|
|
77
77
|
- `Client.upload_volume_file(volume_path, data: bytes)` / `Client.delete_volume_file(volume_path)` -- Unity Catalog volume files via the Files API. Delete treats a 404 as success (idempotent). Both raise a plain `RuntimeError` (message only) on failure.
|
|
78
|
-
- `Client.execute_streamed(statement, *, catalog=None, schema=None, parameters=None, total_timeout_s=None)` -- like `execute()`, but an async iterator yielding the `HEARTBEAT` singleton while waiting on Databricks instead of blocking silently (bridge e.g. an SSE connection through a slow warehouse cold start), then a `ResultSet` exactly once. Raises if `total_timeout_s` elapses first.
|
|
79
|
-
- `ResultSet.fetchall_arrow_streamed(*, total_timeout_s=None)` -- same idea for the chunk-download phase: yields `HEARTBEAT` while pulling chunks, then a `Table` exactly once.
|
|
80
78
|
- `write_ipc_stream(stream, buf)` -- free function; writes any object implementing `__arrow_c_stream__` (a `Table` from this crate, arro3, pyarrow, ...) as uncompressed Arrow-IPC stream bytes to a Python file-like object. No dependency needed regardless of the input's origin.
|
|
81
79
|
- `read_ipc_stream(data: bytes) -> Table` -- free function; the exact inverse of `write_ipc_stream`, parsing raw Arrow-IPC stream bytes back into a `Table`. No dependency needed regardless of where the bytes came from -- backs `arrowbricks.ReplayableArrowChunk`, which needs to re-parse the same cached bytes on every `__arrow_c_stream__` call.
|
|
82
80
|
- `HEARTBEAT` -- module-level singleton; compare with `is`, e.g. `if item is _core.HEARTBEAT: ...`.
|
|
@@ -90,7 +88,8 @@ directly, no pandas/pyarrow conversion step, no extra dependency:
|
|
|
90
88
|
```python
|
|
91
89
|
import duckdb
|
|
92
90
|
|
|
93
|
-
|
|
91
|
+
result_set = await client.execute("SELECT * FROM my_catalog.my_schema.my_table")
|
|
92
|
+
result = await result_set.fetchall_arrow()
|
|
94
93
|
print(duckdb.sql("SELECT count(*) FROM result").fetchall())
|
|
95
94
|
```
|
|
96
95
|
|
|
@@ -27,7 +27,8 @@ async def main() -> None:
|
|
|
27
27
|
warehouse_id=os.environ["DATABRICKS_WAREHOUSE_ID"],
|
|
28
28
|
token=os.environ["DATABRICKS_TOKEN"],
|
|
29
29
|
)
|
|
30
|
-
|
|
30
|
+
result_set = await client.execute("SELECT * FROM my_catalog.my_schema.my_table LIMIT 1000000")
|
|
31
|
+
result = await result_set.fetchall_arrow() # noqa: F841 -- read by DuckDB's replacement scan via local variable name, not a literal Python reference
|
|
31
32
|
|
|
32
33
|
# `result` is queryable by variable name -- DuckDB imports it zero-copy,
|
|
33
34
|
# no pandas/pyarrow conversion step in between.
|
|
@@ -56,6 +56,15 @@ pub struct ColumnDescription {
|
|
|
56
56
|
pub name: String,
|
|
57
57
|
#[serde(default)]
|
|
58
58
|
pub type_name: Option<String>,
|
|
59
|
+
/// Only present for `type_name == "DECIMAL"` -- needed by
|
|
60
|
+
/// `json_convert`'s INLINE/JSON_ARRAY-to-Arrow conversion to build a
|
|
61
|
+
/// correctly-scaled `Decimal128Array`. Absent (and unused) for every
|
|
62
|
+
/// other type, including on the Arrow-IPC path, which gets its decimal
|
|
63
|
+
/// precision/scale from the IPC schema itself, not from here.
|
|
64
|
+
#[serde(default)]
|
|
65
|
+
pub type_precision: Option<u8>,
|
|
66
|
+
#[serde(default)]
|
|
67
|
+
pub type_scale: Option<i8>,
|
|
59
68
|
}
|
|
60
69
|
|
|
61
70
|
#[derive(Deserialize, Default)]
|
|
@@ -119,6 +128,14 @@ struct ResultLinkBody {
|
|
|
119
128
|
struct ResultBody {
|
|
120
129
|
#[serde(default)]
|
|
121
130
|
external_links: Vec<ResultLinkBody>,
|
|
131
|
+
/// Only present for `disposition: INLINE` + `format: JSON_ARRAY` -- each
|
|
132
|
+
/// row a `Vec<Option<String>>` (every non-null value a string,
|
|
133
|
+
/// Databricks' own JSON_ARRAY contract, same as `execute_json_statement`'s
|
|
134
|
+
/// normal `EXTERNAL_LINKS`+`JSON_ARRAY` chunks). Consumed by
|
|
135
|
+
/// `json_convert::json_array_to_record_batch` on the `prefer_inline`
|
|
136
|
+
/// fast path.
|
|
137
|
+
#[serde(default)]
|
|
138
|
+
data_array: Option<Vec<Vec<Option<String>>>>,
|
|
122
139
|
}
|
|
123
140
|
|
|
124
141
|
#[derive(Deserialize)]
|
|
@@ -197,7 +214,18 @@ impl ApiError {
|
|
|
197
214
|
/// (excluding `is_connect()`/`is_timeout()`, already handled) also covers
|
|
198
215
|
/// hyper's own "connection closed before message completed" pooled-
|
|
199
216
|
/// connection-reuse race, which fires before any body is even read and
|
|
200
|
-
/// is equally safe to retry on an idempotent request.
|
|
217
|
+
/// is equally safe to retry on an idempotent request. Unlike `is_decode()`
|
|
218
|
+
/// (hit for real on a 400-chunk fetch and reproduced with a directed
|
|
219
|
+
/// test), this specific race is reasoned from reqwest/hyper's own source,
|
|
220
|
+
/// not independently reproduced -- a directed attempt to force it
|
|
221
|
+
/// self-healed 20/20 times (hyper's own idle-connection health check
|
|
222
|
+
/// evidently detects a closed peer and opens a fresh connection before
|
|
223
|
+
/// ever handing the dead one out for reuse, at least under a simple,
|
|
224
|
+
/// low-concurrency test). Kept anyway since it's safe regardless (scoped
|
|
225
|
+
/// to idempotent requests only) and the window may still be reachable
|
|
226
|
+
/// under real production concurrency even though a quick local test
|
|
227
|
+
/// couldn't force it -- treat as a plausible defensive measure, not a
|
|
228
|
+
/// confirmed fix for an observed failure.
|
|
201
229
|
fn from_reqwest(e: reqwest::Error, idempotent: bool) -> Self {
|
|
202
230
|
// `is_decode()` (reqwest's `Kind::Decode`) isn't only content-decoding
|
|
203
231
|
// -- `Response::bytes()`/`.text()`/`.json()` also wrap a body read
|
|
@@ -219,8 +247,18 @@ impl ApiError {
|
|
|
219
247
|
}
|
|
220
248
|
}
|
|
221
249
|
|
|
222
|
-
|
|
223
|
-
|
|
250
|
+
/// `idempotent` only gates the 5xx case -- 401/403/408/429 mean the
|
|
251
|
+
/// request was rejected before any processing started (auth failure,
|
|
252
|
+
/// rate limit, client-side timeout), safe to retry regardless of method.
|
|
253
|
+
/// A 5xx is murkier: it usually means the same, but it can also mean the
|
|
254
|
+
/// backend already accepted and started the statement before some later
|
|
255
|
+
/// failure (e.g. a gateway timeout) produced the 5xx anyway -- found in
|
|
256
|
+
/// code review that this call was unconditionally transient even for the
|
|
257
|
+
/// statement-submit POST, bypassing the exact idempotency reasoning
|
|
258
|
+
/// `from_reqwest`'s `idempotent` param exists for (retrying that POST
|
|
259
|
+
/// risks a second, duplicate execution of arbitrary caller SQL).
|
|
260
|
+
fn from_status(status: StatusCode, body: &str, idempotent: bool) -> Self {
|
|
261
|
+
let transient = matches!(status.as_u16(), 401 | 403 | 408 | 429) || (idempotent && status.is_server_error());
|
|
224
262
|
Self {
|
|
225
263
|
message: format!("HTTP {status}: {body}"),
|
|
226
264
|
transient,
|
|
@@ -271,6 +309,23 @@ pub struct StatementSubmitResult {
|
|
|
271
309
|
pub compressed: bool,
|
|
272
310
|
}
|
|
273
311
|
|
|
312
|
+
/// What `execute_arrow_statement_prefer_inline` gets you -- either the whole
|
|
313
|
+
/// result inline as raw JSON_ARRAY rows (every non-null value a string,
|
|
314
|
+
/// same contract as `execute_json_statement`; converting these into a real
|
|
315
|
+
/// `RecordBatch` is `pipeline.rs`'s job via `json_convert`, not this
|
|
316
|
+
/// Arrow-agnostic module's -- see this file's own module doc comment), or a
|
|
317
|
+
/// normal `StatementSubmitResult` to fetch chunks for exactly as if
|
|
318
|
+
/// `prefer_inline` had never been asked for. See that function's own doc
|
|
319
|
+
/// comment for when each happens.
|
|
320
|
+
pub enum InlineOrExternal {
|
|
321
|
+
Inline {
|
|
322
|
+
statement_id: String,
|
|
323
|
+
rows: Vec<Vec<Option<String>>>,
|
|
324
|
+
columns: Vec<ColumnDescription>,
|
|
325
|
+
},
|
|
326
|
+
External(StatementSubmitResult),
|
|
327
|
+
}
|
|
328
|
+
|
|
274
329
|
#[derive(Debug)]
|
|
275
330
|
pub struct ChunkItem {
|
|
276
331
|
/// `Bytes` (not `Vec<u8>`) so the bytes reqwest already received off the
|
|
@@ -316,14 +371,22 @@ fn decompress_lz4_frame(compressed: &Bytes) -> Result<Bytes, ApiError> {
|
|
|
316
371
|
// small lower bound.
|
|
317
372
|
let mut out = Vec::with_capacity(compressed.len() * 4);
|
|
318
373
|
let mut decoder = lz4_flex::frame::FrameDecoder::new(&compressed[..]);
|
|
319
|
-
|
|
320
|
-
|
|
374
|
+
// Terminate on the *reader* being exhausted, not on "output stopped
|
|
375
|
+
// growing" -- found in code review that a frame which happens to decode
|
|
376
|
+
// to zero bytes (a real, valid LZ4 Frame shape: header + immediate
|
|
377
|
+
// EndMark) makes one `read_to_end` call return `Ok(0)` for that frame
|
|
378
|
+
// without erroring and without necessarily advancing into the next
|
|
379
|
+
// frame yet, which the old `out.len() == before` check read as "no more
|
|
380
|
+
// frames" -- silently dropping every subsequent concatenated frame with
|
|
381
|
+
// no error at all. Same silent-truncation shape as the original
|
|
382
|
+
// multi-frame bug this loop exists to fix. Looping while the reader
|
|
383
|
+
// still has bytes left (regardless of whether the last call grew `out`)
|
|
384
|
+
// is the correct fix -- verified against a zero-content frame sandwiched
|
|
385
|
+
// between two real ones, see `decompress_lz4_frame_survives_a_zero_content_frame_in_the_middle`.
|
|
386
|
+
while !decoder.get_ref().is_empty() {
|
|
321
387
|
decoder
|
|
322
388
|
.read_to_end(&mut out)
|
|
323
389
|
.map_err(|e| ApiError::permanent(format!("LZ4 frame decompress failed: {e}")))?;
|
|
324
|
-
if out.len() == before {
|
|
325
|
-
break;
|
|
326
|
-
}
|
|
327
390
|
}
|
|
328
391
|
Ok(Bytes::from(out))
|
|
329
392
|
}
|
|
@@ -499,7 +562,7 @@ impl DbClient {
|
|
|
499
562
|
let status = resp.status();
|
|
500
563
|
let text = resp.text().await.map_err(|e| ApiError::from_reqwest(e, idempotent))?;
|
|
501
564
|
if !status.is_success() {
|
|
502
|
-
return Err(ApiError::from_status(status, &text));
|
|
565
|
+
return Err(ApiError::from_status(status, &text, idempotent));
|
|
503
566
|
}
|
|
504
567
|
serde_json::from_str::<T>(&text).map_err(|e| ApiError::permanent(format!("bad JSON body: {e}")))
|
|
505
568
|
})
|
|
@@ -530,7 +593,7 @@ impl DbClient {
|
|
|
530
593
|
let status = resp.status();
|
|
531
594
|
if !status.is_success() {
|
|
532
595
|
let text = resp.text().await.unwrap_or_default();
|
|
533
|
-
return Err(ApiError::from_status(status, &text));
|
|
596
|
+
return Err(ApiError::from_status(status, &text, true));
|
|
534
597
|
}
|
|
535
598
|
let bytes = resp.bytes().await.map_err(|e| ApiError::from_reqwest(e, true))?;
|
|
536
599
|
if !compressed {
|
|
@@ -595,6 +658,101 @@ impl DbClient {
|
|
|
595
658
|
.await
|
|
596
659
|
}
|
|
597
660
|
|
|
661
|
+
/// Tries `disposition: INLINE` + `format: JSON_ARRAY` first -- for a
|
|
662
|
+
/// small result, Databricks embeds the whole result directly in this
|
|
663
|
+
/// same submit/poll response (`result.data_array`), skipping the
|
|
664
|
+
/// separate chunk-resolution-and-blob-fetch round trip the normal
|
|
665
|
+
/// `EXTERNAL_LINKS` path always needs. Confirmed against a real
|
|
666
|
+
/// workspace: exceeding INLINE's byte limit (26,214,400 bytes / 25MiB)
|
|
667
|
+
/// fails the statement cleanly with a specific, matchable error
|
|
668
|
+
/// message -- never silent truncation -- and `INLINE_OR_EXTERNAL_LINKS`
|
|
669
|
+
/// ("HYBRID", which would let the server choose per-query) returned a
|
|
670
|
+
/// clean "not a supported disposition" 400 on that same workspace, so
|
|
671
|
+
/// isn't used here. `format` must be `JSON_ARRAY` for `INLINE` --
|
|
672
|
+
/// Databricks rejects `INLINE`+`ARROW_STREAM` outright (also confirmed).
|
|
673
|
+
/// This module stays Arrow-agnostic on purpose (see its own module doc
|
|
674
|
+
/// comment) -- `InlineOrExternal::Inline` carries the raw JSON_ARRAY
|
|
675
|
+
/// rows straight through; converting them into a `RecordBatch` (and
|
|
676
|
+
/// falling back to a fresh `EXTERNAL_LINKS` submission if that
|
|
677
|
+
/// conversion hits a column type it doesn't handle) is `pipeline.rs`'s
|
|
678
|
+
/// job via `json_convert`, same division of responsibility as every
|
|
679
|
+
/// other decode step in this crate.
|
|
680
|
+
///
|
|
681
|
+
/// Falls back to a **fresh, independent** `execute_arrow_statement` call
|
|
682
|
+
/// (not a retry of this same submission) whenever INLINE doesn't pan
|
|
683
|
+
/// out at the HTTP/statement level: the byte limit was exceeded, or
|
|
684
|
+
/// `data_array` is unexpectedly absent despite SUCCEEDED. This is a
|
|
685
|
+
/// second, distinct statement execution, not a retry of a possibly-
|
|
686
|
+
/// already-run one, so it carries none of the double-execution risk
|
|
687
|
+
/// that made POST retries unsafe elsewhere in this file -- the first
|
|
688
|
+
/// attempt's outcome is fully known (it reached a terminal state)
|
|
689
|
+
/// before the second one is ever submitted. A genuine query error (bad
|
|
690
|
+
/// SQL, permission denied) propagates immediately instead, same as the
|
|
691
|
+
/// normal path -- no reason to mask it behind a pointless second
|
|
692
|
+
/// attempt.
|
|
693
|
+
///
|
|
694
|
+
/// Not the default -- opt-in only (`prefer_inline` on the Python-facing
|
|
695
|
+
/// `execute()`/`Cursor.execute()`), since a caller who doesn't expect a
|
|
696
|
+
/// small result pays for two full statement executions on the (common,
|
|
697
|
+
/// for them) fallback path instead of one.
|
|
698
|
+
pub async fn execute_arrow_statement_prefer_inline(
|
|
699
|
+
&self,
|
|
700
|
+
statement: &str,
|
|
701
|
+
catalog: Option<&str>,
|
|
702
|
+
schema: Option<&str>,
|
|
703
|
+
parameters: Option<Value>,
|
|
704
|
+
) -> Result<InlineOrExternal, ApiError> {
|
|
705
|
+
let mut body = json!({
|
|
706
|
+
"warehouse_id": self.warehouse_id,
|
|
707
|
+
"statement": statement,
|
|
708
|
+
"disposition": "INLINE",
|
|
709
|
+
"format": "JSON_ARRAY",
|
|
710
|
+
"wait_timeout": self.wait_timeout,
|
|
711
|
+
"on_wait_timeout": "CONTINUE",
|
|
712
|
+
});
|
|
713
|
+
if let Some(c) = catalog {
|
|
714
|
+
body["catalog"] = json!(c);
|
|
715
|
+
}
|
|
716
|
+
if let Some(s) = schema {
|
|
717
|
+
body["schema"] = json!(s);
|
|
718
|
+
}
|
|
719
|
+
if let Some(p) = parameters.clone() {
|
|
720
|
+
body["parameters"] = p;
|
|
721
|
+
}
|
|
722
|
+
|
|
723
|
+
let outcome = self.submit_and_poll(body).await;
|
|
724
|
+
let data = match outcome {
|
|
725
|
+
Ok(d) => d,
|
|
726
|
+
Err(e) if e.message.contains("Inline byte limit exceeded") => {
|
|
727
|
+
return self
|
|
728
|
+
.execute_arrow_statement(statement, catalog, schema, parameters)
|
|
729
|
+
.await
|
|
730
|
+
.map(InlineOrExternal::External);
|
|
731
|
+
}
|
|
732
|
+
Err(e) => return Err(e),
|
|
733
|
+
};
|
|
734
|
+
|
|
735
|
+
let manifest = data.manifest.unwrap_or_default();
|
|
736
|
+
let columns = manifest.schema.map(|s| s.columns).unwrap_or_default();
|
|
737
|
+
let data_array = data.result.and_then(|r| r.data_array);
|
|
738
|
+
match data_array {
|
|
739
|
+
Some(rows) => Ok(InlineOrExternal::Inline {
|
|
740
|
+
statement_id: data.statement_id,
|
|
741
|
+
rows,
|
|
742
|
+
columns,
|
|
743
|
+
}),
|
|
744
|
+
// No data_array despite SUCCEEDED -- shouldn't happen given a
|
|
745
|
+
// non-error status, but this crate never guesses at a missing
|
|
746
|
+
// field; a fresh EXTERNAL_LINKS submission is exactly as safe as
|
|
747
|
+
// the byte-limit fallback above (a distinct statement, not a
|
|
748
|
+
// retry of this one).
|
|
749
|
+
None => self
|
|
750
|
+
.execute_arrow_statement(statement, catalog, schema, parameters)
|
|
751
|
+
.await
|
|
752
|
+
.map(InlineOrExternal::External),
|
|
753
|
+
}
|
|
754
|
+
}
|
|
755
|
+
|
|
598
756
|
/// Like `execute_arrow_statement`, fixed to JSON_ARRAY -- each fetched
|
|
599
757
|
/// chunk's bytes are then a JSON array of rows, each row itself an array
|
|
600
758
|
/// of values where every non-null value is a *string* regardless of its
|
|
@@ -613,6 +771,41 @@ impl DbClient {
|
|
|
613
771
|
.await
|
|
614
772
|
}
|
|
615
773
|
|
|
774
|
+
/// Shared by `execute_statement` and `execute_arrow_statement_prefer_inline`:
|
|
775
|
+
/// POST the statement, poll until a terminal state, and turn FAILED/
|
|
776
|
+
/// CANCELED into an `Err` -- everything both callers need before they
|
|
777
|
+
/// diverge on how to interpret a SUCCEEDED response's `result`/`manifest`.
|
|
778
|
+
async fn submit_and_poll(&self, body: Value) -> Result<StatementResponseBody, ApiError> {
|
|
779
|
+
self.ensure_warehouse_running().await?;
|
|
780
|
+
let url = format!("{}/api/2.0/sql/statements", self.host);
|
|
781
|
+
let mut data: StatementResponseBody = self.authed_json(reqwest::Method::POST, &url, Some(&body)).await?;
|
|
782
|
+
|
|
783
|
+
while !matches!(
|
|
784
|
+
data.status.state.as_str(),
|
|
785
|
+
"SUCCEEDED" | "FAILED" | "CANCELED" | "CLOSED"
|
|
786
|
+
) {
|
|
787
|
+
tokio::time::sleep(POLL_INTERVAL).await;
|
|
788
|
+
let poll_url = format!("{}/api/2.0/sql/statements/{}", self.host, data.statement_id);
|
|
789
|
+
data = self.authed_json(reqwest::Method::GET, &poll_url, None).await?;
|
|
790
|
+
}
|
|
791
|
+
|
|
792
|
+
match data.status.state.as_str() {
|
|
793
|
+
"FAILED" => {
|
|
794
|
+
let err = data.status.error.unwrap_or(StatementErrorBody {
|
|
795
|
+
error_code: None,
|
|
796
|
+
message: None,
|
|
797
|
+
});
|
|
798
|
+
Err(ApiError::permanent(format!(
|
|
799
|
+
"Databricks statement failed [{}]: {}",
|
|
800
|
+
err.error_code.as_deref().unwrap_or(""),
|
|
801
|
+
err.message.as_deref().unwrap_or(""),
|
|
802
|
+
)))
|
|
803
|
+
}
|
|
804
|
+
"CANCELED" => Err(ApiError::permanent("Databricks statement was canceled")),
|
|
805
|
+
_ => Ok(data),
|
|
806
|
+
}
|
|
807
|
+
}
|
|
808
|
+
|
|
616
809
|
async fn execute_statement(
|
|
617
810
|
&self,
|
|
618
811
|
statement: &str,
|
|
@@ -649,35 +842,7 @@ impl DbClient {
|
|
|
649
842
|
body["parameters"] = p;
|
|
650
843
|
}
|
|
651
844
|
|
|
652
|
-
self.
|
|
653
|
-
let url = format!("{}/api/2.0/sql/statements", self.host);
|
|
654
|
-
let mut data: StatementResponseBody = self.authed_json(reqwest::Method::POST, &url, Some(&body)).await?;
|
|
655
|
-
|
|
656
|
-
while !matches!(
|
|
657
|
-
data.status.state.as_str(),
|
|
658
|
-
"SUCCEEDED" | "FAILED" | "CANCELED" | "CLOSED"
|
|
659
|
-
) {
|
|
660
|
-
tokio::time::sleep(POLL_INTERVAL).await;
|
|
661
|
-
let poll_url = format!("{}/api/2.0/sql/statements/{}", self.host, data.statement_id);
|
|
662
|
-
data = self.authed_json(reqwest::Method::GET, &poll_url, None).await?;
|
|
663
|
-
}
|
|
664
|
-
|
|
665
|
-
match data.status.state.as_str() {
|
|
666
|
-
"FAILED" => {
|
|
667
|
-
let err = data.status.error.unwrap_or(StatementErrorBody {
|
|
668
|
-
error_code: None,
|
|
669
|
-
message: None,
|
|
670
|
-
});
|
|
671
|
-
return Err(ApiError::permanent(format!(
|
|
672
|
-
"Databricks statement failed [{}]: {}",
|
|
673
|
-
err.error_code.as_deref().unwrap_or(""),
|
|
674
|
-
err.message.as_deref().unwrap_or(""),
|
|
675
|
-
)));
|
|
676
|
-
}
|
|
677
|
-
"CANCELED" => return Err(ApiError::permanent("Databricks statement was canceled")),
|
|
678
|
-
_ => {}
|
|
679
|
-
}
|
|
680
|
-
|
|
845
|
+
let data = self.submit_and_poll(body).await?;
|
|
681
846
|
let manifest = data.manifest.unwrap_or_default();
|
|
682
847
|
let compressed = manifest.result_compression.as_deref() == Some("LZ4_FRAME");
|
|
683
848
|
// `Vec` per index, not a plain map entry -- see `ChunkMeta::pre_resolved_links`'s
|
|
@@ -837,7 +1002,7 @@ impl DbClient {
|
|
|
837
1002
|
let status = resp.status();
|
|
838
1003
|
if !status.is_success() {
|
|
839
1004
|
let text = resp.text().await.unwrap_or_default();
|
|
840
|
-
return Err(ApiError::from_status(status, &text));
|
|
1005
|
+
return Err(ApiError::from_status(status, &text, true));
|
|
841
1006
|
}
|
|
842
1007
|
Ok(())
|
|
843
1008
|
})
|
|
@@ -866,7 +1031,7 @@ impl DbClient {
|
|
|
866
1031
|
return Ok(());
|
|
867
1032
|
}
|
|
868
1033
|
let text = resp.text().await.unwrap_or_default();
|
|
869
|
-
Err(ApiError::from_status(status, &text))
|
|
1034
|
+
Err(ApiError::from_status(status, &text, true))
|
|
870
1035
|
})
|
|
871
1036
|
.await
|
|
872
1037
|
}
|
|
@@ -932,6 +1097,44 @@ mod tests {
|
|
|
932
1097
|
assert_eq!(decompressed, Bytes::from(expected));
|
|
933
1098
|
}
|
|
934
1099
|
|
|
1100
|
+
/// Regression test for a bug found in code review: a real, valid LZ4
|
|
1101
|
+
/// Frame that happens to decode to zero bytes (a header immediately
|
|
1102
|
+
/// followed by an EndMark -- a legal frame shape, not malformed input)
|
|
1103
|
+
/// makes `read_to_end` return `Ok(0)` for that frame without erroring.
|
|
1104
|
+
/// The old loop read "output didn't grow" as "no more frames" and
|
|
1105
|
+
/// stopped there, silently dropping every frame concatenated after the
|
|
1106
|
+
/// empty one. `decompress_lz4_frame` must keep going as long as the
|
|
1107
|
+
/// underlying reader still has bytes left, not just as long as output
|
|
1108
|
+
/// keeps growing.
|
|
1109
|
+
#[test]
|
|
1110
|
+
fn decompress_lz4_frame_survives_a_zero_content_frame_in_the_middle() {
|
|
1111
|
+
use std::io::Write;
|
|
1112
|
+
|
|
1113
|
+
fn compress_one_frame(data: &[u8]) -> Vec<u8> {
|
|
1114
|
+
let mut encoder = lz4_flex::frame::FrameEncoder::new(Vec::new());
|
|
1115
|
+
encoder.write_all(data).unwrap();
|
|
1116
|
+
encoder.finish().unwrap()
|
|
1117
|
+
}
|
|
1118
|
+
|
|
1119
|
+
let part_a = b"the quick brown fox jumps over the lazy dog ".repeat(50);
|
|
1120
|
+
let part_b = b"pack my box with five dozen liquor jugs ".repeat(50);
|
|
1121
|
+
let mut concatenated_frames = Vec::new();
|
|
1122
|
+
concatenated_frames.extend(compress_one_frame(&part_a));
|
|
1123
|
+
concatenated_frames.extend(compress_one_frame(b"")); // real frame, zero content
|
|
1124
|
+
concatenated_frames.extend(compress_one_frame(&part_b));
|
|
1125
|
+
|
|
1126
|
+
let decompressed = decompress_lz4_frame(&Bytes::from(concatenated_frames)).unwrap();
|
|
1127
|
+
|
|
1128
|
+
let mut expected = Vec::new();
|
|
1129
|
+
expected.extend_from_slice(&part_a);
|
|
1130
|
+
expected.extend_from_slice(&part_b);
|
|
1131
|
+
assert_eq!(
|
|
1132
|
+
decompressed,
|
|
1133
|
+
Bytes::from(expected),
|
|
1134
|
+
"the frame after the zero-content one must not be silently dropped"
|
|
1135
|
+
);
|
|
1136
|
+
}
|
|
1137
|
+
|
|
935
1138
|
fn ok_task() -> tokio::task::JoinHandle<Result<(), ApiError>> {
|
|
936
1139
|
tokio::spawn(async { Ok(()) })
|
|
937
1140
|
}
|