arrowbricks 3.2.0__tar.gz → 5.0.0__tar.gz

This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
Files changed (50) hide show
  1. {arrowbricks-3.2.0 → arrowbricks-5.0.0}/PKG-INFO +10 -1
  2. {arrowbricks-3.2.0 → arrowbricks-5.0.0}/README.md +9 -0
  3. {arrowbricks-3.2.0 → arrowbricks-5.0.0}/pyproject.toml +4 -7
  4. {arrowbricks-3.2.0 → arrowbricks-5.0.0}/rust/arrowbricks_core/Cargo.lock +3 -136
  5. {arrowbricks-3.2.0 → arrowbricks-5.0.0}/rust/arrowbricks_core/Cargo.toml +9 -19
  6. {arrowbricks-3.2.0 → arrowbricks-5.0.0}/rust/arrowbricks_core/README.md +5 -6
  7. arrowbricks-5.0.0/rust/arrowbricks_core/src/arrow_ffi.rs +106 -0
  8. {arrowbricks-3.2.0 → arrowbricks-5.0.0}/rust/arrowbricks_core/src/client/sea.rs +3 -0
  9. {arrowbricks-3.2.0 → arrowbricks-5.0.0}/rust/arrowbricks_core/src/client.rs +23 -9
  10. {arrowbricks-3.2.0 → arrowbricks-5.0.0}/rust/arrowbricks_core/src/lib.rs +8 -12
  11. {arrowbricks-3.2.0 → arrowbricks-5.0.0}/rust/arrowbricks_core/src/pipeline/ndjson.rs +19 -0
  12. {arrowbricks-3.2.0 → arrowbricks-5.0.0}/rust/arrowbricks_core/src/pipeline/sea.rs +1 -1
  13. {arrowbricks-3.2.0 → arrowbricks-5.0.0}/rust/arrowbricks_core/src/pipeline/thrift_exec.rs +3 -2
  14. {arrowbricks-3.2.0 → arrowbricks-5.0.0}/rust/arrowbricks_core/tests/wiremock_pipeline.rs +46 -0
  15. {arrowbricks-3.2.0 → arrowbricks-5.0.0}/rust/arrowbricks_core/tests/wiremock_thrift.rs +120 -0
  16. {arrowbricks-3.2.0 → arrowbricks-5.0.0}/rust/arrowbricks_core/tests_py/test_ipc_stream.py +38 -0
  17. {arrowbricks-3.2.0 → arrowbricks-5.0.0}/rust/arrowbricks_core/tests_py/test_thrift.py +5 -4
  18. {arrowbricks-3.2.0 → arrowbricks-5.0.0}/src/arrowbricks/__init__.py +4 -0
  19. {arrowbricks-3.2.0 → arrowbricks-5.0.0}/src/arrowbricks/_core.pyi +16 -4
  20. {arrowbricks-3.2.0 → arrowbricks-5.0.0}/src/arrowbricks/_streaming.py +8 -2
  21. {arrowbricks-3.2.0 → arrowbricks-5.0.0}/src/arrowbricks/cursor.py +3 -0
  22. {arrowbricks-3.2.0 → arrowbricks-5.0.0}/LICENSE +0 -0
  23. {arrowbricks-3.2.0 → arrowbricks-5.0.0}/rust/arrowbricks_core/.gitignore +0 -0
  24. {arrowbricks-3.2.0 → arrowbricks-5.0.0}/rust/arrowbricks_core/examples/duckdb_query.py +0 -0
  25. {arrowbricks-3.2.0 → arrowbricks-5.0.0}/rust/arrowbricks_core/examples/fastapi_sse.py +0 -0
  26. {arrowbricks-3.2.0 → arrowbricks-5.0.0}/rust/arrowbricks_core/proptest-regressions/json_convert.txt +0 -0
  27. {arrowbricks-3.2.0 → arrowbricks-5.0.0}/rust/arrowbricks_core/rustfmt.toml +0 -0
  28. {arrowbricks-3.2.0 → arrowbricks-5.0.0}/rust/arrowbricks_core/src/client/download.rs +0 -0
  29. {arrowbricks-3.2.0 → arrowbricks-5.0.0}/rust/arrowbricks_core/src/client/error.rs +0 -0
  30. {arrowbricks-3.2.0 → arrowbricks-5.0.0}/rust/arrowbricks_core/src/client/model.rs +0 -0
  31. {arrowbricks-3.2.0 → arrowbricks-5.0.0}/rust/arrowbricks_core/src/client/thrift_rpc.rs +0 -0
  32. {arrowbricks-3.2.0 → arrowbricks-5.0.0}/rust/arrowbricks_core/src/client/volume.rs +0 -0
  33. {arrowbricks-3.2.0 → arrowbricks-5.0.0}/rust/arrowbricks_core/src/heartbeat.rs +0 -0
  34. {arrowbricks-3.2.0 → arrowbricks-5.0.0}/rust/arrowbricks_core/src/json_convert.rs +0 -0
  35. {arrowbricks-3.2.0 → arrowbricks-5.0.0}/rust/arrowbricks_core/src/pipeline/reorder.rs +0 -0
  36. {arrowbricks-3.2.0 → arrowbricks-5.0.0}/rust/arrowbricks_core/src/pipeline/stats.rs +0 -0
  37. {arrowbricks-3.2.0 → arrowbricks-5.0.0}/rust/arrowbricks_core/src/pipeline/test_support.rs +0 -0
  38. {arrowbricks-3.2.0 → arrowbricks-5.0.0}/rust/arrowbricks_core/src/pipeline.rs +0 -0
  39. {arrowbricks-3.2.0 → arrowbricks-5.0.0}/rust/arrowbricks_core/src/thrift.rs +0 -0
  40. {arrowbricks-3.2.0 → arrowbricks-5.0.0}/rust/arrowbricks_core/tests/common/mod.rs +0 -0
  41. {arrowbricks-3.2.0 → arrowbricks-5.0.0}/rust/arrowbricks_core/tests/wiremock_volume_files.rs +0 -0
  42. {arrowbricks-3.2.0 → arrowbricks-5.0.0}/rust/arrowbricks_core/tests_py/conftest.py +0 -0
  43. {arrowbricks-3.2.0 → arrowbricks-5.0.0}/rust/arrowbricks_core/tests_py/test_parameters.py +0 -0
  44. {arrowbricks-3.2.0 → arrowbricks-5.0.0}/rust/arrowbricks_core/tests_py/test_stream_ndjson_lines.py +0 -0
  45. {arrowbricks-3.2.0 → arrowbricks-5.0.0}/rust/arrowbricks_core/tests_py/test_streaming.py +0 -0
  46. {arrowbricks-3.2.0 → arrowbricks-5.0.0}/rust/arrowbricks_core/tests_py/test_token_provider.py +0 -0
  47. {arrowbricks-3.2.0 → arrowbricks-5.0.0}/rust/arrowbricks_core/tests_py/test_volume_files.py +0 -0
  48. {arrowbricks-3.2.0 → arrowbricks-5.0.0}/rust/arrowbricks_core/tests_py/thrift_mock.py +0 -0
  49. {arrowbricks-3.2.0 → arrowbricks-5.0.0}/src/arrowbricks/client.py +0 -0
  50. {arrowbricks-3.2.0 → arrowbricks-5.0.0}/src/arrowbricks/py.typed +0 -0
@@ -1,6 +1,6 @@
1
1
  Metadata-Version: 2.4
2
2
  Name: arrowbricks
3
- Version: 3.2.0
3
+ Version: 5.0.0
4
4
  Requires-Dist: arro3-core>=0.8 ; extra == 'arro3'
5
5
  Provides-Extra: arro3
6
6
  License-File: LICENSE
@@ -212,6 +212,15 @@ except ArrowbricksError:
212
212
 
213
213
  ## Arrow vs. row-tuple fetches
214
214
 
215
+ From 4.0, Arrow fetches and `ReplayableArrowChunk.to_table()` return
216
+ `arrowbricks._core.Table`. It exposes `num_rows`, `num_columns`,
217
+ `column_names`, `len(table)` and the Arrow capsule interfaces. For methods
218
+ such as `.column()`, `.schema` or `.to_batches()`, first convert with
219
+ `arro3.core.Table.from_arrow(table)`, `pyarrow.table(table)` or
220
+ `polars.from_arrow(table)`. DuckDB can consume it directly.
221
+ `write_ipc_stream` requires `__arrow_c_stream__`; wrap an array-only producer
222
+ in a table before passing it in.
223
+
215
224
  Everything above works with zero dependencies installed *except* row-tuple fetches. `fetchall_arrow`/`fetchmany_arrow` return an Arrow table straight from the Rust core -- the faster path if your code can consume Arrow directly (DuckDB, pyarrow, polars, a Parquet writer, ...):
216
225
 
217
226
  ```python
@@ -200,6 +200,15 @@ except ArrowbricksError:
200
200
 
201
201
  ## Arrow vs. row-tuple fetches
202
202
 
203
+ From 4.0, Arrow fetches and `ReplayableArrowChunk.to_table()` return
204
+ `arrowbricks._core.Table`. It exposes `num_rows`, `num_columns`,
205
+ `column_names`, `len(table)` and the Arrow capsule interfaces. For methods
206
+ such as `.column()`, `.schema` or `.to_batches()`, first convert with
207
+ `arro3.core.Table.from_arrow(table)`, `pyarrow.table(table)` or
208
+ `polars.from_arrow(table)`. DuckDB can consume it directly.
209
+ `write_ipc_stream` requires `__arrow_c_stream__`; wrap an array-only producer
210
+ in a table before passing it in.
211
+
203
212
  Everything above works with zero dependencies installed *except* row-tuple fetches. `fetchall_arrow`/`fetchmany_arrow` return an Arrow table straight from the Rust core -- the faster path if your code can consume Arrow directly (DuckDB, pyarrow, polars, a Parquet writer, ...):
204
213
 
205
214
  ```python
@@ -1,6 +1,6 @@
1
1
  [project]
2
2
  name = "arrowbricks"
3
- version = "3.2.0"
3
+ version = "5.0.0"
4
4
  description = "Runs SQL against a Databricks SQL warehouse via the Statement Execution API and hands you the result as Arrow -- a DB-API-ish Cursor (fetchone/fetchmany/fetchall/fetchall_arrow) or NDJSON streaming. Rust/PyO3 core throughout -- zero required runtime dependencies."
5
5
  readme = "README.md"
6
6
  license = "MIT"
@@ -15,12 +15,8 @@ requires-python = ">=3.11"
15
15
  dependencies = []
16
16
 
17
17
  [project.optional-dependencies]
18
- # pyo3-arrow's own Table.column()/schema methods are designed to hand back
19
- # the caller's *real* arro3.core objects for further arro3-ecosystem
20
- # interop (not a value pyo3-arrow can materialize itself), so
21
- # fetchone/fetchmany/fetchall need this installed. `pip install
22
- # arrowbricks[arro3]` -- or handle rows as Arrow via fetchall_arrow/
23
- # fetchmany_arrow instead, which don't need it at all.
18
+ # Row materialization converts our capsule table through arro3-core.
19
+ # Arrow fetches and streaming do not need this optional dependency.
24
20
  arro3 = ["arro3-core>=0.8"]
25
21
 
26
22
  [project.urls]
@@ -80,6 +76,7 @@ dev = [
80
76
 
81
77
  [tool.pytest.ini_options]
82
78
  asyncio_mode = "auto"
79
+ asyncio_default_fixture_loop_scope = "function"
83
80
  # Bare `pytest`/`pytest -q` (no explicit path) covers tests/ -- including
84
81
  # tests/arro3_free (nested here on purpose; see its own conftest.py for why
85
82
  # that doesn't pull arro3 into its collection) -- but NOT
@@ -81,7 +81,6 @@ dependencies = [
81
81
  "atoi",
82
82
  "base64 0.23.1",
83
83
  "chrono",
84
- "comfy-table",
85
84
  "half",
86
85
  "lexical-core",
87
86
  "num-traits",
@@ -178,7 +177,7 @@ dependencies = [
178
177
 
179
178
  [[package]]
180
179
  name = "arrowbricks_core"
181
- version = "3.2.0"
180
+ version = "5.0.0"
182
181
  dependencies = [
183
182
  "arrow-array",
184
183
  "arrow-buffer",
@@ -193,7 +192,6 @@ dependencies = [
193
192
  "memchr",
194
193
  "proptest",
195
194
  "pyo3",
196
- "pyo3-arrow",
197
195
  "pyo3-async-runtimes",
198
196
  "pythonize",
199
197
  "reqwest",
@@ -303,9 +301,7 @@ source = "registry+https://github.com/rust-lang/crates.io-index"
303
301
  checksum = "1aa79e62e7697b8e29b513a68abacf485adcd1fe8284a4316c5ae868e6633327"
304
302
  dependencies = [
305
303
  "iana-time-zone",
306
- "js-sys",
307
304
  "num-traits",
308
- "wasm-bindgen",
309
305
  "windows-link",
310
306
  ]
311
307
 
@@ -329,16 +325,6 @@ dependencies = [
329
325
  "memchr",
330
326
  ]
331
327
 
332
- [[package]]
333
- name = "comfy-table"
334
- version = "7.2.2"
335
- source = "registry+https://github.com/rust-lang/crates.io-index"
336
- checksum = "958c5d6ecf1f214b4c2bbbbf6ab9523a864bd136dcf71a7e8904799acfe1ad47"
337
- dependencies = [
338
- "unicode-segmentation",
339
- "unicode-width",
340
- ]
341
-
342
328
  [[package]]
343
329
  name = "const-random"
344
330
  version = "0.1.18"
@@ -889,7 +875,7 @@ dependencies = [
889
875
  "jni-sys",
890
876
  "log",
891
877
  "simd_cesu8",
892
- "thiserror 2.0.20",
878
+ "thiserror",
893
879
  "walkdir",
894
880
  "windows-link",
895
881
  ]
@@ -1039,16 +1025,6 @@ dependencies = [
1039
1025
  "twox-hash",
1040
1026
  ]
1041
1027
 
1042
- [[package]]
1043
- name = "matrixmultiply"
1044
- version = "0.3.11"
1045
- source = "registry+https://github.com/rust-lang/crates.io-index"
1046
- checksum = "3f607c237553f086e7043417a51df26b2eb899d3caff94e6a67592ff992fedc7"
1047
- dependencies = [
1048
- "autocfg",
1049
- "rawpointer",
1050
- ]
1051
-
1052
1028
  [[package]]
1053
1029
  name = "memchr"
1054
1030
  version = "2.8.3"
@@ -1066,21 +1042,6 @@ dependencies = [
1066
1042
  "windows-sys 0.61.2",
1067
1043
  ]
1068
1044
 
1069
- [[package]]
1070
- name = "ndarray"
1071
- version = "0.17.2"
1072
- source = "registry+https://github.com/rust-lang/crates.io-index"
1073
- checksum = "520080814a7a6b4a6e9070823bb24b4531daac8c4627e08ba5de8c5ef2f2752d"
1074
- dependencies = [
1075
- "matrixmultiply",
1076
- "num-complex",
1077
- "num-integer",
1078
- "num-traits",
1079
- "portable-atomic",
1080
- "portable-atomic-util",
1081
- "rawpointer",
1082
- ]
1083
-
1084
1045
  [[package]]
1085
1046
  name = "num-bigint"
1086
1047
  version = "0.5.1"
@@ -1129,23 +1090,6 @@ dependencies = [
1129
1090
  "libc",
1130
1091
  ]
1131
1092
 
1132
- [[package]]
1133
- name = "numpy"
1134
- version = "0.29.0"
1135
- source = "registry+https://github.com/rust-lang/crates.io-index"
1136
- checksum = "6a5b15d63a5ff39e378daed0e1340d3a5964703ea9712eb09a0dc66fade996f4"
1137
- dependencies = [
1138
- "half",
1139
- "libc",
1140
- "ndarray",
1141
- "num-complex",
1142
- "num-integer",
1143
- "num-traits",
1144
- "pyo3",
1145
- "pyo3-build-config",
1146
- "rustc-hash",
1147
- ]
1148
-
1149
1093
  [[package]]
1150
1094
  name = "once_cell"
1151
1095
  version = "1.21.4"
@@ -1194,15 +1138,6 @@ version = "1.14.0"
1194
1138
  source = "registry+https://github.com/rust-lang/crates.io-index"
1195
1139
  checksum = "3d20d5497ef88037a52ff98267d066e7f11fcc5e99bbfbd58a42336193aacec3"
1196
1140
 
1197
- [[package]]
1198
- name = "portable-atomic-util"
1199
- version = "0.2.7"
1200
- source = "registry+https://github.com/rust-lang/crates.io-index"
1201
- checksum = "c2a106d1259c23fac8e543272398ae0e3c0b8d33c88ed73d0cc71b0f1d902618"
1202
- dependencies = [
1203
- "portable-atomic",
1204
- ]
1205
-
1206
1141
  [[package]]
1207
1142
  name = "potential_utf"
1208
1143
  version = "0.1.5"
@@ -1255,9 +1190,6 @@ version = "0.29.2"
1255
1190
  source = "registry+https://github.com/rust-lang/crates.io-index"
1256
1191
  checksum = "4688ddedf473e32662b9b067670129a8afb8c18e351482c70d62ba4a88171e8b"
1257
1192
  dependencies = [
1258
- "chrono",
1259
- "chrono-tz",
1260
- "indexmap",
1261
1193
  "libc",
1262
1194
  "once_cell",
1263
1195
  "portable-atomic",
@@ -1266,27 +1198,6 @@ dependencies = [
1266
1198
  "pyo3-macros",
1267
1199
  ]
1268
1200
 
1269
- [[package]]
1270
- name = "pyo3-arrow"
1271
- version = "0.19.0"
1272
- source = "registry+https://github.com/rust-lang/crates.io-index"
1273
- checksum = "3d5ddf226a2dbf7607570d0657c2bf6fbe299208368b2914f3dd7e7ba0b57688"
1274
- dependencies = [
1275
- "arrow-array",
1276
- "arrow-buffer",
1277
- "arrow-cast",
1278
- "arrow-data",
1279
- "arrow-schema",
1280
- "arrow-select",
1281
- "chrono",
1282
- "chrono-tz",
1283
- "half",
1284
- "indexmap",
1285
- "numpy",
1286
- "pyo3",
1287
- "thiserror 1.0.69",
1288
- ]
1289
-
1290
1201
  [[package]]
1291
1202
  name = "pyo3-async-runtimes"
1292
1203
  version = "0.29.0"
@@ -1413,12 +1324,6 @@ dependencies = [
1413
1324
  "rand_core",
1414
1325
  ]
1415
1326
 
1416
- [[package]]
1417
- name = "rawpointer"
1418
- version = "0.2.1"
1419
- source = "registry+https://github.com/rust-lang/crates.io-index"
1420
- checksum = "60a357793950651c4ed0f3f52338f53b2f809f32d83a07f72909fa13e4c6c1e3"
1421
-
1422
1327
  [[package]]
1423
1328
  name = "regex"
1424
1329
  version = "1.13.1"
@@ -1499,12 +1404,6 @@ dependencies = [
1499
1404
  "windows-sys 0.52.0",
1500
1405
  ]
1501
1406
 
1502
- [[package]]
1503
- name = "rustc-hash"
1504
- version = "2.1.3"
1505
- source = "registry+https://github.com/rust-lang/crates.io-index"
1506
- checksum = "6b1e7f9a428571be2dc5bc0505c13fb6bf936822b894ec87abf8a08a4e51742d"
1507
-
1508
1407
  [[package]]
1509
1408
  name = "rustc_version"
1510
1409
  version = "0.4.1"
@@ -1850,33 +1749,13 @@ dependencies = [
1850
1749
  "windows-sys 0.61.2",
1851
1750
  ]
1852
1751
 
1853
- [[package]]
1854
- name = "thiserror"
1855
- version = "1.0.69"
1856
- source = "registry+https://github.com/rust-lang/crates.io-index"
1857
- checksum = "b6aaf5339b578ea85b50e080feb250a3e8ae8cfcdff9a461c9ec2904bc923f52"
1858
- dependencies = [
1859
- "thiserror-impl 1.0.69",
1860
- ]
1861
-
1862
1752
  [[package]]
1863
1753
  name = "thiserror"
1864
1754
  version = "2.0.20"
1865
1755
  source = "registry+https://github.com/rust-lang/crates.io-index"
1866
1756
  checksum = "ec86235f5fcc2a73650310756d2ac5b138a5780bbbdfae3eeccec992c435ba4f"
1867
1757
  dependencies = [
1868
- "thiserror-impl 2.0.20",
1869
- ]
1870
-
1871
- [[package]]
1872
- name = "thiserror-impl"
1873
- version = "1.0.69"
1874
- source = "registry+https://github.com/rust-lang/crates.io-index"
1875
- checksum = "4fee6c4efc90059e10f81e6d42c60a18f76588c3d74cb83a0b242a2b6c7504c1"
1876
- dependencies = [
1877
- "proc-macro2",
1878
- "quote",
1879
- "syn 2.0.119",
1758
+ "thiserror-impl",
1880
1759
  ]
1881
1760
 
1882
1761
  [[package]]
@@ -2047,18 +1926,6 @@ version = "1.0.24"
2047
1926
  source = "registry+https://github.com/rust-lang/crates.io-index"
2048
1927
  checksum = "e6e4313cd5fcd3dad5cafa179702e2b244f760991f45397d14d4ebf38247da75"
2049
1928
 
2050
- [[package]]
2051
- name = "unicode-segmentation"
2052
- version = "1.13.3"
2053
- source = "registry+https://github.com/rust-lang/crates.io-index"
2054
- checksum = "c6f5d3c3b1bf09027a88a6bc961fc00497d651009560b5463668dc81b0fa87a8"
2055
-
2056
- [[package]]
2057
- name = "unicode-width"
2058
- version = "0.2.2"
2059
- source = "registry+https://github.com/rust-lang/crates.io-index"
2060
- checksum = "b4ac048d71ede7ee76d585517add45da530660ef4390e49b098733c6e897f254"
2061
-
2062
1929
  [[package]]
2063
1930
  name = "untrusted"
2064
1931
  version = "0.9.0"
@@ -1,6 +1,6 @@
1
1
  [package]
2
2
  name = "arrowbricks_core"
3
- version = "3.2.0"
3
+ version = "5.0.0"
4
4
  edition = "2024"
5
5
  readme = "README.md"
6
6
 
@@ -11,7 +11,10 @@ crate-type = ["cdylib", "rlib"]
11
11
  [dependencies]
12
12
  # Depend on the components we use. The umbrella `arrow` crate also builds
13
13
  # arithmetic, string, and row kernels even with default-features disabled.
14
- arrow-array = "59.1.0"
14
+ # `ffi` backs src/arrow_ffi.rs, this crate's own Arrow PyCapsule bridge.
15
+ # `chrono-tz` lets arrow-json format Databricks' named-zone timestamps
16
+ # ("Etc/UTC"); pyo3-arrow used to enable it implicitly.
17
+ arrow-array = { version = "59.1.0", features = ["ffi", "chrono-tz"] }
15
18
  arrow-buffer = "59.1.0"
16
19
  arrow-ipc = "59.1.0"
17
20
  arrow-schema = "59.1.0"
@@ -20,15 +23,14 @@ bytes = "1.12.1"
20
23
  # Already in the dependency graph; vectorized newline search in NDJSON output.
21
24
  memchr = "2"
22
25
  # Already an unconditional transitive dependency of arrow-cast (pulled in by
23
- # pyo3-arrow) at this exact version -- declared directly here too so
26
+ # arrow-json) at this exact version -- declared directly here too so
24
27
  # json_convert.rs can decode BINARY columns without hand-rolling base64. No
25
28
  # new code ships that wasn't already linked in (feature unification, same
26
29
  # reasoning as chrono/rustls below).
27
30
  base64 = "0.23"
28
- # Already an unconditional transitive dependency of pyo3-arrow (see its own
29
- # comment below) -- declared directly here too so json_convert.rs can parse
30
- # DATE/TIMESTAMP/TIMESTAMP_NTZ strings without hand-rolling calendar
31
- # arithmetic. No new code ships that wasn't already linked in.
31
+ # Already an unconditional transitive dependency of arrow-array -- declared
32
+ # directly here too so json_convert.rs can parse DATE/TIMESTAMP/TIMESTAMP_NTZ
33
+ # strings without hand-rolling calendar arithmetic. No new code ships that wasn't already linked in.
32
34
  chrono = { version = "0.4", default-features = false, features = ["std"] }
33
35
  # Decompresses cloud-fetch chunk bytes when the server honors our
34
36
  # `result_compression: LZ4_FRAME` request (see client.rs's execute_statement)
@@ -36,12 +38,6 @@ chrono = { version = "0.4", default-features = false, features = ["std"] }
36
38
  # Pure Rust, no C toolchain dependency -- unlike reqwest's TLS stack below.
37
39
  lz4_flex = { version = "0.14", default-features = false, features = ["frame", "std"] }
38
40
  pyo3 = { version = "0.29.1", features = ["abi3-py311"] }
39
- # pyo3-arrow's only optional feature is buffer_protocol (on by default),
40
- # which we don't use -- numpy/chrono/chrono-tz/arrow-cast(prettyprint) below
41
- # it are unconditional dependencies of the crate itself, not features, so
42
- # they ship regardless; not worth dropping pyo3-arrow over that (reimplementing
43
- # the Arrow C Data Interface ourselves is a much bigger risk than the MBs saved).
44
- pyo3-arrow = { version = "0.19.0", default-features = false }
45
41
  pyo3-async-runtimes = { version = "0.29.0", features = ["tokio-runtime"] }
46
42
  pythonize = "0.29.0"
47
43
  # reqwest's own `rustls` feature is hardcoded to `__rustls-aws-lc-rs`, no way
@@ -135,9 +131,3 @@ opt-level = "s"
135
131
  strip = true
136
132
  lto = true
137
133
  codegen-units = 1
138
-
139
- [profile.release.package.pyo3-arrow]
140
- # This bridge is dominated by interoperability helpers. Optimizing it for
141
- # size saves ~32 KiB on macOS arm64; the IPC decoder keeps opt-level="s".
142
- # Replay and live-query benchmarks are recorded in benchmarks/2026-09-06.md.
143
- opt-level = "z"
@@ -63,12 +63,11 @@ asyncio.run(main())
63
63
  ```
64
64
 
65
65
  `table` implements the [Arrow C Data Interface](https://arrow.apache.org/docs/format/CDataInterface.html)
66
- (`__arrow_c_stream__`/`__arrow_c_array__`) -- any consumer that speaks that
67
- protocol (DuckDB, pyarrow, arro3) can import it with zero copies, no extra
68
- dependency needed. Note: `table.column(i)`/`table.schema` (materializing
69
- native Python values) are `pyo3-arrow` methods designed to hand back the
70
- caller's own *real* `arro3.core` objects -- they need `arro3-core` installed,
71
- unlike the table object itself.
66
+ (`__arrow_c_stream__`/`__arrow_c_schema__`) -- any consumer that speaks that
67
+ protocol (DuckDB, pyarrow, polars, arro3) can import it with zero copies, no
68
+ extra dependency needed. The table itself only exposes `num_rows`,
69
+ `num_columns`, `column_names` and `len()`; to read values, import it into an
70
+ Arrow library first (e.g. `arro3.core.Table.from_arrow(table)`).
72
71
 
73
72
  ## API
74
73
 
@@ -0,0 +1,106 @@
1
+ //! Arrow PyCapsule interface, both directions, on arrow-rs's own
2
+ //! `ffi_stream` -- replaces `pyo3-arrow`, whose `Table.__arrow_c_stream__`
3
+ //! honors `requested_schema` via `arrow_cast::cast` and so kept the whole cast
4
+ //! kernel (~1.8 MiB, a third of `.text`) linked in for a feature no caller of
5
+ //! this crate used.
6
+
7
+ use std::ffi::CStr;
8
+
9
+ use arrow_array::ffi::FFI_ArrowSchema;
10
+ use arrow_array::ffi_stream::{ArrowArrayStreamReader, FFI_ArrowArrayStream};
11
+ use arrow_array::{RecordBatch, RecordBatchIterator};
12
+ use arrow_schema::SchemaRef;
13
+ use pyo3::exceptions::PyRuntimeError;
14
+ use pyo3::prelude::*;
15
+ use pyo3::types::{PyCapsule, PyCapsuleMethods};
16
+
17
+ const STREAM_CAPSULE: &CStr = c"arrow_array_stream";
18
+ const SCHEMA_CAPSULE: &CStr = c"arrow_schema";
19
+
20
+ /// An in-memory Arrow table. Consume it through `__arrow_c_stream__`
21
+ /// (arro3, pyarrow, polars, DuckDB, ...); the attributes are just enough to
22
+ /// inspect a result without importing another Arrow library.
23
+ #[pyclass(name = "Table", module = "arrowbricks._core", frozen)]
24
+ pub struct PyTable {
25
+ batches: Vec<RecordBatch>,
26
+ schema: SchemaRef,
27
+ }
28
+
29
+ impl PyTable {
30
+ pub fn try_new(batches: Vec<RecordBatch>, schema: SchemaRef) -> PyResult<Self> {
31
+ // A stream advertises one schema for every batch. Exporting different
32
+ // buffer layouts under it can make consumers misinterpret memory.
33
+ // Like the previous bridge, ignore metadata and nullability differences.
34
+ if batches.iter().any(|batch| {
35
+ let fields = batch.schema_ref().fields();
36
+ fields.len() != schema.fields().len()
37
+ || fields.iter().zip(schema.fields()).any(|(actual, expected)| {
38
+ actual.name() != expected.name() || !actual.data_type().equals_datatype(expected.data_type())
39
+ })
40
+ }) {
41
+ return Err(PyRuntimeError::new_err("All batches must have same schema"));
42
+ }
43
+ Ok(Self { batches, schema })
44
+ }
45
+ }
46
+
47
+ #[pymethods]
48
+ impl PyTable {
49
+ #[getter]
50
+ fn num_rows(&self) -> usize {
51
+ self.batches.iter().map(RecordBatch::num_rows).sum()
52
+ }
53
+
54
+ #[getter]
55
+ fn num_columns(&self) -> usize {
56
+ self.schema.fields().len()
57
+ }
58
+
59
+ #[getter]
60
+ fn column_names(&self) -> Vec<String> {
61
+ self.schema.fields().iter().map(|f| f.name().clone()).collect()
62
+ }
63
+
64
+ fn __len__(&self) -> usize {
65
+ self.num_rows()
66
+ }
67
+
68
+ fn __repr__(&self) -> String {
69
+ format!(
70
+ "arrowbricks.Table(num_rows={}, num_columns={})",
71
+ self.num_rows(),
72
+ self.num_columns()
73
+ )
74
+ }
75
+
76
+ /// `requested_schema` is ignored, as the PyCapsule interface allows: the
77
+ /// consumer checks the schema it actually gets.
78
+ #[pyo3(signature = (requested_schema=None))]
79
+ fn __arrow_c_stream__<'py>(
80
+ &self,
81
+ py: Python<'py>,
82
+ requested_schema: Option<Bound<'py, PyAny>>,
83
+ ) -> PyResult<Bound<'py, PyCapsule>> {
84
+ let _ = requested_schema;
85
+ let reader = RecordBatchIterator::new(self.batches.clone().into_iter().map(Ok), self.schema.clone());
86
+ PyCapsule::new_with_value(py, FFI_ArrowArrayStream::new(Box::new(reader)), STREAM_CAPSULE)
87
+ }
88
+
89
+ fn __arrow_c_schema__<'py>(&self, py: Python<'py>) -> PyResult<Bound<'py, PyCapsule>> {
90
+ let schema =
91
+ FFI_ArrowSchema::try_from(self.schema.as_ref()).map_err(|e| PyRuntimeError::new_err(e.to_string()))?;
92
+ PyCapsule::new_with_value(py, schema, SCHEMA_CAPSULE)
93
+ }
94
+ }
95
+
96
+ /// Imports any object implementing `__arrow_c_stream__`.
97
+ pub fn import_stream(obj: &Bound<'_, PyAny>) -> PyResult<ArrowArrayStreamReader> {
98
+ let capsule = obj.call_method0("__arrow_c_stream__")?.cast_into::<PyCapsule>()?;
99
+ let ptr = capsule
100
+ .pointer_checked(Some(STREAM_CAPSULE))?
101
+ .cast::<FFI_ArrowArrayStream>();
102
+ // SAFETY: the capsule name check guarantees an `ArrowArrayStream`;
103
+ // `from_raw` moves it out and marks the capsule's copy released, so the
104
+ // capsule destructor does not release it a second time.
105
+ unsafe { ArrowArrayStreamReader::from_raw(ptr.as_ptr()) }.map_err(|e| PyRuntimeError::new_err(e.to_string()))
106
+ }
@@ -381,6 +381,9 @@ impl DbClient {
381
381
  session_id,
382
382
  };
383
383
  let result = self.submit_and_poll_inner(body, stats).await;
384
+ if result.is_ok() {
385
+ self.note_warehouse_running();
386
+ }
384
387
  checkin.finish(result.is_ok());
385
388
  result
386
389
  }
@@ -217,13 +217,18 @@ pub const MAX_SPLIT_PARTS: usize = 8;
217
217
  /// silently could.
218
218
  const DEFAULT_CHUNK_FETCH_CONCURRENCY: usize = 64;
219
219
 
220
- /// How long the Thrift path polls `GetOperationStatus` when a statement
221
- /// doesn't finish within its `getDirectResults` budget (see
222
- /// `execute_lazy_thrift`) -- shorter than SEA's `POLL_INTERVAL` (2s) since
223
- /// this path exists specifically to be fast for small/quick queries; a
224
- /// query slow enough to need many polls pays a modest, bounded amount of
225
- /// extra round trips either way.
226
- pub(crate) const THRIFT_POLL_INTERVAL: Duration = Duration::from_millis(200);
220
+ /// Sleep before the `attempt`-th (0-based) re-poll of Thrift
221
+ /// `GetOperationStatus` when a statement doesn't finish within its
222
+ /// `getDirectResults` budget (see `execute_lazy_thrift`): a short ramp
223
+ /// (10/25/50/100ms), then 200ms steady. A fixed 200ms made a statement that
224
+ /// finished just after the first poll pay a full 200ms of pure sleep; the
225
+ /// ramp costs at most four extra cheap RPCs before settling at the old
226
+ /// steady rate. Shorter than SEA's `POLL_INTERVAL` (2s) since this path
227
+ /// exists specifically to be fast for small/quick queries.
228
+ pub(crate) fn thrift_poll_delay(attempt: usize) -> Duration {
229
+ const RAMP_MS: [u64; 5] = [10, 25, 50, 100, 200];
230
+ Duration::from_millis(RAMP_MS[attempt.min(RAMP_MS.len() - 1)])
231
+ }
227
232
 
228
233
  /// Request hints on `TExecuteStatementReq.getDirectResults`/`TFetchResultsReq` --
229
234
  /// how much of the result the server should try to hand back in one RPC.
@@ -644,6 +649,15 @@ impl DbClient {
644
649
  .await
645
650
  }
646
651
 
652
+ /// Refreshes `ensure_warehouse_running`'s RUNNING cache. Also called once a
653
+ /// statement reaches a successful terminal state (SEA's `submit_and_poll`,
654
+ /// Thrift's `submit_and_await_thrift_statement`): a statement that just ran
655
+ /// proves the warehouse is up as well as the REST GET does, so steady
656
+ /// traffic with gaps under the TTL never pays that GET again.
657
+ pub(crate) fn note_warehouse_running(&self) {
658
+ *self.warehouse_confirmed_running_at.lock().unwrap() = Some(Instant::now());
659
+ }
660
+
647
661
  /// Shared by both protocols -- plain REST against `/api/2.0/sql/warehouses/{id}`,
648
662
  /// nothing SEA- or Thrift-specific about it (see AGENTS.md for why the
649
663
  /// Thrift path didn't always call this).
@@ -660,7 +674,7 @@ impl DbClient {
660
674
  let url = format!("{}/api/2.0/sql/warehouses/{}", self.host, self.warehouse_id);
661
675
  let data: WarehouseStatusBody = self.authed_json(reqwest::Method::GET, &url, None, None).await?;
662
676
  if data.state == "RUNNING" {
663
- *self.warehouse_confirmed_running_at.lock().unwrap() = Some(Instant::now());
677
+ self.note_warehouse_running();
664
678
  return Ok(());
665
679
  }
666
680
  if data.state == "STOPPED" {
@@ -673,7 +687,7 @@ impl DbClient {
673
687
  tokio::time::sleep(POLL_INTERVAL).await;
674
688
  let data: WarehouseStatusBody = self.authed_json(reqwest::Method::GET, &url, None, None).await?;
675
689
  if data.state == "RUNNING" {
676
- *self.warehouse_confirmed_running_at.lock().unwrap() = Some(Instant::now());
690
+ self.note_warehouse_running();
677
691
  return Ok(());
678
692
  }
679
693
  }
@@ -1,3 +1,4 @@
1
+ mod arrow_ffi;
1
2
  pub mod client;
2
3
  pub mod heartbeat;
3
4
  pub mod json_convert;
@@ -12,11 +13,10 @@ use pyo3::exceptions::{PyRuntimeError, PyStopAsyncIteration, PyValueError};
12
13
  use pyo3::prelude::*;
13
14
  use pyo3::sync::PyOnceLock;
14
15
  use pyo3::types::PyBytes;
15
- use pyo3_arrow::PyTable;
16
- use pyo3_arrow::input::AnyRecordBatch;
17
16
  use pyo3_async_runtimes::TaskLocals;
18
17
  use tokio::sync::Mutex as AsyncMutex;
19
18
 
19
+ use arrow_ffi::PyTable;
20
20
  use client::{
21
21
  ApiError, ApiErrorKind, CancelHandle, DbClient, EventSink, Protocol, QueryStatsAccumulator, QueryStatsData,
22
22
  TokenFuture, TokenProvider,
@@ -33,9 +33,8 @@ use pipeline::{NdjsonStream, ResultStream};
33
33
  #[pyfunction]
34
34
  #[pyo3(signature = (stream, buf))]
35
35
  fn write_ipc_stream(py: Python<'_>, stream: Bound<'_, PyAny>, buf: Bound<'_, PyAny>) -> PyResult<()> {
36
- let any_rb: AnyRecordBatch = stream.extract()?;
37
- let mut reader = any_rb.into_reader()?;
38
- let schema = reader.schema();
36
+ let mut reader = arrow_ffi::import_stream(&stream)?;
37
+ let schema = arrow_array::RecordBatchReader::schema(&reader);
39
38
  let mut ipc_buf: Vec<u8> = Vec::new();
40
39
  {
41
40
  let mut writer = arrow_ipc::writer::StreamWriter::try_new(&mut ipc_buf, &schema)
@@ -70,7 +69,7 @@ fn read_ipc_stream(data: Bound<'_, PyBytes>) -> PyResult<PyTable> {
70
69
  let (batches, schema) = py
71
70
  .detach(|| pipeline::decode_ipc_stream(&blob))
72
71
  .map_err(|e| PyRuntimeError::new_err(e.message))?;
73
- PyTable::try_new(batches, schema).map_err(|e| PyRuntimeError::new_err(e.to_string()))
72
+ PyTable::try_new(batches, schema)
74
73
  }
75
74
 
76
75
  /// Wraps a `PyErr` raised by the caller's own `token_provider` callable (its
@@ -804,7 +803,7 @@ fn column_pairs(columns: &[client::ColumnDescription]) -> Vec<(String, Option<St
804
803
  /// schema rather than a schema-less `Table`.
805
804
  fn batches_to_pytable(batches: Vec<RecordBatch>, schema: Option<SchemaRef>) -> PyResult<PyTable> {
806
805
  let schema = schema.unwrap_or_else(|| Arc::new(Schema::empty()));
807
- PyTable::try_new(batches, schema).map_err(|e| PyRuntimeError::new_err(e.to_string()))
806
+ PyTable::try_new(batches, schema)
808
807
  }
809
808
 
810
809
  #[pymethods]
@@ -854,11 +853,7 @@ impl PyResultSet {
854
853
  /// `Cursor.description`'s shape. `None` before any fetch (the caller
855
854
  /// should fall back to `columns`, the manifest-based pre-fetch
856
855
  /// estimate). Computed directly from the decoded `arrow_schema::Schema`
857
- /// here rather than via a returned `Table`'s own `.schema` property --
858
- /// that property specifically requires a *real* `arro3.core` install to
859
- /// construct its return value (by pyo3-arrow's own design, so callers
860
- /// get their own runtime's Schema type back), which would reintroduce
861
- /// exactly the dependency this crate's callers don't have.
856
+ /// here, so no Arrow library is needed to read it.
862
857
  fn schema<'py>(&self, py: Python<'py>) -> PyResult<Bound<'py, PyAny>> {
863
858
  let inner = self.inner.clone();
864
859
  pyo3_async_runtimes::tokio::future_into_py(py, async move {
@@ -1065,6 +1060,7 @@ impl PyNdjsonStreamIter {
1065
1060
  fn _core(m: &Bound<'_, PyModule>) -> PyResult<()> {
1066
1061
  m.add_function(wrap_pyfunction!(write_ipc_stream, m)?)?;
1067
1062
  m.add_function(wrap_pyfunction!(read_ipc_stream, m)?)?;
1063
+ m.add_class::<PyTable>()?;
1068
1064
  m.add_class::<PyDbClient>()?;
1069
1065
  m.add_class::<PyResultSet>()?;
1070
1066
  m.add_class::<PyHeartbeat>()?;
@@ -576,6 +576,25 @@ mod tests {
576
576
  assert_eq!(string_lines[3], r#"{"id":4,"value":1.5}"#); // finite values untouched
577
577
  }
578
578
 
579
+ /// Databricks sends TIMESTAMP as `Timestamp(us, "Etc/UTC")`. A named zone
580
+ /// needs arrow-array's `chrono-tz` feature, which pyo3-arrow used to
581
+ /// enable implicitly; without it this errors ("only offset based
582
+ /// timezones supported").
583
+ #[test]
584
+ fn encode_ndjson_lines_handles_a_named_timezone() {
585
+ use arrow_array::TimestampMicrosecondArray;
586
+ use arrow_schema::{Field, Schema, TimeUnit};
587
+ let array = TimestampMicrosecondArray::from(vec![1_767_225_600_000_000]).with_timezone("Etc/UTC");
588
+ let schema = Arc::new(Schema::new(vec![Field::new(
589
+ "ts",
590
+ DataType::Timestamp(TimeUnit::Microsecond, Some("Etc/UTC".into())),
591
+ false,
592
+ )]));
593
+ let batch = RecordBatch::try_new(schema, vec![Arc::new(array)]).unwrap();
594
+ let lines = encode_ndjson_lines(std::slice::from_ref(&batch), false).unwrap();
595
+ assert_eq!(lines[0], r#"{"ts":"2026-01-01T00:00:00Z"}"#);
596
+ }
597
+
579
598
  #[test]
580
599
  fn encode_ndjson_lines_leaves_a_real_null_alone_when_requested() {
581
600
  use arrow_array::{Float64Array, Int64Array};
@@ -465,7 +465,7 @@ pub async fn execute_lazy_prefer_inline(
465
465
 
466
466
  /// Full submit -> poll -> fetch -> reorder -> decode pipeline. Returns the
467
467
  /// assembled batches in logical (chunk_index) order plus their schema, ready
468
- /// to hand to `pyo3_arrow::PyTable` for a zero-copy Arrow C Data Interface
468
+ /// to hand to `arrow_ffi::PyTable` for a zero-copy Arrow C Data Interface
469
469
  /// handoff back to Python (consumable by DuckDB/pyarrow/arro3 directly).
470
470
  ///
471
471
  /// Decode of each reordered chunk is pushed onto `spawn_blocking` rather
@@ -140,7 +140,7 @@ async fn submit_and_await_thrift_statement(
140
140
  stats.set_in_flight(CancelHandle::Thrift {
141
141
  operation: operation.clone(),
142
142
  });
143
- loop {
143
+ for attempt in 0.. {
144
144
  let status = client.thrift_get_operation_status_raw(&operation, stats).await?;
145
145
  if let Some(e) = status.terminal_error() {
146
146
  stats.clear_in_flight();
@@ -150,9 +150,10 @@ async fn submit_and_await_thrift_statement(
150
150
  stats.clear_in_flight();
151
151
  break;
152
152
  }
153
- tokio::time::sleep(crate::client::THRIFT_POLL_INTERVAL).await;
153
+ tokio::time::sleep(crate::client::thrift_poll_delay(attempt)).await;
154
154
  }
155
155
  }
156
+ client.note_warehouse_running();
156
157
 
157
158
  Ok(ThriftStatementReady {
158
159
  operation,
@@ -1187,6 +1187,52 @@ async fn session_is_created_once_and_reused_across_sequential_statements() {
1187
1187
  }
1188
1188
  }
1189
1189
 
1190
+ /// SEA counterpart of `wiremock_thrift.rs`'s cache-refresh test: a SUCCEEDED
1191
+ /// statement refreshes `ensure_warehouse_running`'s cache, so the third of
1192
+ /// three statements ~600ms apart (1s TTL) needs no second warehouse GET.
1193
+ #[tokio::test]
1194
+ async fn successful_statement_refreshes_the_warehouse_running_cache() {
1195
+ let server = MockServer::start().await;
1196
+ mount_warehouse_running(&server).await;
1197
+ Mock::given(method("POST"))
1198
+ .and(path("/api/2.0/sql/sessions"))
1199
+ .respond_with(ResponseTemplate::new(200).set_body_json(json!({"session_id": "sess-0"})))
1200
+ .mount(&server)
1201
+ .await;
1202
+ Mock::given(method("POST"))
1203
+ .and(path("/api/2.0/sql/statements"))
1204
+ .respond_with(ResponseTemplate::new(200).set_body_json(json!({
1205
+ "statement_id": STATEMENT_ID,
1206
+ "status": {"state": "SUCCEEDED"},
1207
+ "manifest": {"chunks": []},
1208
+ })))
1209
+ .mount(&server)
1210
+ .await;
1211
+
1212
+ let client = Arc::new(
1213
+ DbClient::new(&server.uri(), WAREHOUSE_ID, "fake-token")
1214
+ .with_protocol(Protocol::Sea)
1215
+ .with_warehouse_confirmed_running_ttl(1.0),
1216
+ );
1217
+ for i in 0..3 {
1218
+ if i > 0 {
1219
+ tokio::time::sleep(std::time::Duration::from_millis(600)).await;
1220
+ }
1221
+ run_pipeline(client.clone(), "SELECT 1", None, None, None)
1222
+ .await
1223
+ .unwrap();
1224
+ }
1225
+
1226
+ let warehouse_gets = server
1227
+ .received_requests()
1228
+ .await
1229
+ .unwrap()
1230
+ .iter()
1231
+ .filter(|r| r.method == wiremock::http::Method::GET && r.url.path().starts_with("/api/2.0/sql/warehouses/"))
1232
+ .count();
1233
+ assert_eq!(warehouse_gets, 1);
1234
+ }
1235
+
1190
1236
  #[tokio::test]
1191
1237
  async fn session_creation_failure_falls_back_to_catalog_on_the_statement_body() {
1192
1238
  let server = MockServer::start().await;
@@ -1187,6 +1187,126 @@ async fn thrift_session_is_created_once_and_reused_across_sequential_statements(
1187
1187
  );
1188
1188
  }
1189
1189
 
1190
+ /// A successful statement refreshes `ensure_warehouse_running`'s RUNNING
1191
+ /// cache. With a 1s TTL and ~600ms gaps, the third statement is >1s
1192
+ /// after the only warehouse GET but <1s after the second statement
1193
+ /// succeeded -- so exactly one GET only if success refreshes the cache.
1194
+ #[tokio::test]
1195
+ async fn thrift_successful_statement_refreshes_the_warehouse_running_cache() {
1196
+ let server = MockServer::start().await;
1197
+ mount_open_session_always(&server, b"sess").await;
1198
+ mount_close_operation_ok(&server).await;
1199
+
1200
+ let schema = test_schema();
1201
+ let schema_for_mock = schema.clone();
1202
+ Mock::given(method("POST"))
1203
+ .and(path(thrift_path()))
1204
+ .and(IsThriftRpc("ExecuteStatement"))
1205
+ .respond_with(move |_req: &Request| {
1206
+ ResponseTemplate::new(200).set_body_raw(
1207
+ small_execute_statement_success(b"op".to_vec(), &schema_for_mock),
1208
+ "application/x-thrift",
1209
+ )
1210
+ })
1211
+ .mount(&server)
1212
+ .await;
1213
+
1214
+ let client = Arc::new(
1215
+ DbClient::new(&server.uri(), WAREHOUSE_ID, "fake-token")
1216
+ .with_protocol(Protocol::Thrift)
1217
+ .with_warehouse_confirmed_running_ttl(1.0),
1218
+ );
1219
+ for i in 0..3 {
1220
+ if i > 0 {
1221
+ tokio::time::sleep(std::time::Duration::from_millis(600)).await;
1222
+ }
1223
+ let mut stream = execute_lazy_thrift(client.clone(), "SELECT 1", None, None, None)
1224
+ .await
1225
+ .unwrap();
1226
+ stream.fetchall_arrow().await.unwrap();
1227
+ }
1228
+
1229
+ let warehouse_gets = server
1230
+ .received_requests()
1231
+ .await
1232
+ .unwrap()
1233
+ .iter()
1234
+ .filter(|r| r.method == wiremock::http::Method::GET && r.url.path().starts_with("/api/2.0/sql/warehouses/"))
1235
+ .count();
1236
+ assert_eq!(
1237
+ warehouse_gets, 1,
1238
+ "a statement that just succeeded proves the warehouse is running -- no further GET within the TTL"
1239
+ );
1240
+ }
1241
+
1242
+ /// `GetOperationStatus` polling ramps up (10/25/50/100ms, then 200ms) instead
1243
+ /// of a fixed 200ms: two RUNNING polls then FINISHED must cost ~35ms of
1244
+ /// sleep, not the old fixed 400ms.
1245
+ #[tokio::test]
1246
+ async fn thrift_polling_ramps_up_so_a_quick_statement_returns_fast() {
1247
+ let server = MockServer::start().await;
1248
+ mount_open_session_always(&server, b"sess").await;
1249
+ mount_close_operation_ok(&server).await;
1250
+
1251
+ Mock::given(method("POST"))
1252
+ .and(path(thrift_path()))
1253
+ .and(IsThriftRpc("ExecuteStatement"))
1254
+ .respond_with(ResponseTemplate::new(200).set_body_raw(
1255
+ build_execute_statement_resp(b"op-p", b"opsecret-p", None),
1256
+ "application/x-thrift",
1257
+ ))
1258
+ .mount(&server)
1259
+ .await;
1260
+
1261
+ let poll_calls = Arc::new(AtomicUsize::new(0));
1262
+ let poll_calls_for_mock = poll_calls.clone();
1263
+ Mock::given(method("POST"))
1264
+ .and(path(thrift_path()))
1265
+ .and(IsThriftRpc("GetOperationStatus"))
1266
+ .respond_with(move |_req: &Request| {
1267
+ let n = poll_calls_for_mock.fetch_add(1, Ordering::SeqCst);
1268
+ let state = if n < 2 {
1269
+ operation_state::RUNNING
1270
+ } else {
1271
+ operation_state::FINISHED
1272
+ };
1273
+ ResponseTemplate::new(200)
1274
+ .set_body_raw(build_get_operation_status_resp(state, None), "application/x-thrift")
1275
+ })
1276
+ .mount(&server)
1277
+ .await;
1278
+
1279
+ let schema = test_schema();
1280
+ let (schema_bytes, batch_bytes) = build_schema_and_batch_messages(&schema, &[(0, 1)]);
1281
+ Mock::given(method("POST"))
1282
+ .and(path(thrift_path()))
1283
+ .and(IsThriftRpc("FetchResults"))
1284
+ .respond_with(ResponseTemplate::new(200).set_body_raw(
1285
+ build_fetch_results_resp(&FetchSpec {
1286
+ has_more_rows: false,
1287
+ arrow_batches: vec![(batch_bytes[0].clone(), 1)],
1288
+ metadata: Some((false, Some(schema_bytes))),
1289
+ ..Default::default()
1290
+ }),
1291
+ "application/x-thrift",
1292
+ ))
1293
+ .mount(&server)
1294
+ .await;
1295
+
1296
+ let client = thrift_client(&server);
1297
+ let t0 = std::time::Instant::now();
1298
+ let mut stream = execute_lazy_thrift(client, "SELECT 1", None, None, None).await.unwrap();
1299
+ let submit_elapsed = t0.elapsed();
1300
+ let (batches, _) = stream.fetchall_arrow().await.unwrap();
1301
+
1302
+ assert_eq!(poll_calls.load(Ordering::SeqCst), 3);
1303
+ assert_ids_in_order(&batches, 1);
1304
+ assert!(
1305
+ submit_elapsed < std::time::Duration::from_millis(300),
1306
+ "two RUNNING polls should cost ~35ms of ramped sleep, not 2x200ms: took {submit_elapsed:?}"
1307
+ );
1308
+ }
1309
+
1190
1310
  #[tokio::test]
1191
1311
  async fn thrift_pool_exhaustion_falls_back_to_a_throwaway_session_that_still_succeeds() {
1192
1312
  let server = MockServer::start().await;
@@ -164,3 +164,41 @@ def test_read_ipc_stream_allows_another_python_thread_to_run():
164
164
  thread.join(timeout=5)
165
165
  assert result.num_rows == 100_000
166
166
  assert ran_during_decode, "IPC decoding must release the interpreter lock"
167
+
168
+
169
+ def test_table_exports_are_independent_and_outlive_the_table():
170
+ source = core.Table.from_pydict({"id": core.Array([1, 2], type=core.DataType.int64())})
171
+ buf = io.BytesIO()
172
+ arrowbricks_core.write_ipc_stream(source, buf)
173
+ table = arrowbricks_core.read_ipc_stream(buf.getvalue())
174
+ assert len(table) == 2
175
+ assert table.column_names == ["id"]
176
+ # A capsule discarded without a consumer must release only its own reader.
177
+ unused = table.__arrow_c_stream__()
178
+ del unused
179
+ first = table.__arrow_c_stream__(requested_schema=None)
180
+ second = table.__arrow_c_stream__()
181
+ del table, source, buf
182
+ gc.collect()
183
+
184
+ class Export:
185
+ def __init__(self, capsule):
186
+ self.capsule = capsule
187
+
188
+ def __arrow_c_stream__(self, requested_schema=None):
189
+ return self.capsule
190
+
191
+ for capsule in (first, second):
192
+ assert core.Table.from_arrow(Export(capsule))["id"].to_pylist() == [1, 2]
193
+ # Each capsule can be consumed only once; a second import must error.
194
+ with pytest.raises(RuntimeError):
195
+ arrowbricks_core.write_ipc_stream(Export(first), io.BytesIO())
196
+
197
+
198
+ def test_write_ipc_stream_rejects_a_schema_capsule():
199
+ class WrongCapsule:
200
+ def __arrow_c_stream__(self):
201
+ return core.Schema([core.Field("id", core.DataType.int64())]).__arrow_c_schema__()
202
+
203
+ with pytest.raises(ValueError):
204
+ arrowbricks_core.write_ipc_stream(WrongCapsule(), io.BytesIO())
@@ -11,6 +11,7 @@ from __future__ import annotations
11
11
 
12
12
  import pytest
13
13
  import thrift_mock as tm
14
+ from arro3.core import Table
14
15
 
15
16
  from arrowbricks import _core as arrowbricks_core
16
17
 
@@ -51,7 +52,7 @@ async def test_thrift_execute_returns_data_inline_via_direct_results(mock_thrift
51
52
  result = await client.execute("SELECT * FROM t")
52
53
  table = await result.fetchall_arrow()
53
54
  assert table.num_rows == 3
54
- assert table.column("id").to_pylist() == [0, 1, 2]
55
+ assert Table.from_arrow(table).column("id").to_pylist() == [0, 1, 2]
55
56
 
56
57
 
57
58
  @pytest.mark.asyncio
@@ -88,7 +89,7 @@ async def test_thrift_multi_batch_fetch_preserves_order(mock_thrift_server):
88
89
  result = await client.execute("SELECT * FROM t ORDER BY id")
89
90
  table = await result.fetchall_arrow()
90
91
  assert table.num_rows == total_rows
91
- assert table.column("id").to_pylist() == list(range(total_rows))
92
+ assert Table.from_arrow(table).column("id").to_pylist() == list(range(total_rows))
92
93
 
93
94
 
94
95
  @pytest.mark.asyncio
@@ -133,7 +134,7 @@ async def test_thrift_lz4_compressed_result_links_are_decompressed(mock_thrift_s
133
134
  result = await client.execute("SELECT * FROM t")
134
135
  table = await result.fetchall_arrow()
135
136
  assert table.num_rows == 4
136
- assert table.column("id").to_pylist() == [0, 1, 2, 3]
137
+ assert Table.from_arrow(table).column("id").to_pylist() == [0, 1, 2, 3]
137
138
 
138
139
 
139
140
  @pytest.mark.asyncio
@@ -222,4 +223,4 @@ async def test_thrift_is_the_default_protocol_when_omitted(mock_thrift_server):
222
223
  client = arrowbricks_core.Client(host=server.host, warehouse_id=WAREHOUSE_ID, token="fake")
223
224
  result = await client.execute("SELECT * FROM t")
224
225
  table = await result.fetchall_arrow()
225
- assert table.column("id").to_pylist() == [0, 1]
226
+ assert Table.from_arrow(table).column("id").to_pylist() == [0, 1]
@@ -1,3 +1,7 @@
1
+ # PEP 810 (Python 3.15+): defer loading these submodules until a name from
2
+ # them is first used. Older Pythons ignore this list and import eagerly.
3
+ __lazy_modules__ = ["arrowbricks._streaming", "arrowbricks.client", "arrowbricks.cursor"]
4
+
1
5
  from ._core import ArrowbricksError, AuthError, QueryStats, StatementError, TransientError
2
6
  from ._streaming import (
3
7
  HEARTBEAT,
@@ -2,7 +2,19 @@ from collections.abc import Awaitable, Callable
2
2
  from typing import Any, BinaryIO, Literal
3
3
 
4
4
  def write_ipc_stream(stream: Any, buf: BinaryIO) -> None: ... # stream: anything implementing __arrow_c_stream__
5
- def read_ipc_stream(data: bytes) -> Any: ... # Arrow table (__arrow_c_stream__)
5
+
6
+ class Table:
7
+ @property
8
+ def num_rows(self) -> int: ...
9
+ @property
10
+ def num_columns(self) -> int: ...
11
+ @property
12
+ def column_names(self) -> list[str]: ...
13
+ def __len__(self) -> int: ...
14
+ def __arrow_c_stream__(self, requested_schema: object | None = None) -> object: ...
15
+ def __arrow_c_schema__(self) -> object: ...
16
+
17
+ def read_ipc_stream(data: bytes) -> Table: ...
6
18
 
7
19
  class _Heartbeat:
8
20
  def __repr__(self) -> str: ...
@@ -49,14 +61,14 @@ class ResultSet:
49
61
  # for Cursor.description-style compatibility.
50
62
  columns: list[tuple[str, str | None]]
51
63
 
52
- async def fetchmany_arrow(self, n: int) -> Any: ... # Arrow table (__arrow_c_stream__)
53
- async def fetchall_arrow(self) -> Any: ... # Arrow table (__arrow_c_stream__)
64
+ async def fetchmany_arrow(self, n: int) -> Table: ...
65
+ async def fetchall_arrow(self) -> Table: ...
54
66
  def fetchall_arrow_streamed(self, total_timeout_s: float | None = None) -> FetchallArrowStreamedIter: ...
55
67
  async def schema(self) -> list[tuple[str, str]] | None: ... # real schema, known after >=1 fetch
56
68
 
57
69
  class FetchallArrowStreamedIter:
58
70
  def __aiter__(self) -> FetchallArrowStreamedIter: ...
59
- async def __anext__(self) -> Any: ... # Arrow table (__arrow_c_stream__) | _Heartbeat
71
+ async def __anext__(self) -> Table | _Heartbeat: ...
60
72
 
61
73
  class NdjsonStreamIter:
62
74
  def __aiter__(self) -> NdjsonStreamIter: ...
@@ -11,14 +11,16 @@ Python-side Arrow-to-JSON conversion step at all.
11
11
 
12
12
  from __future__ import annotations
13
13
 
14
- import asyncio
15
14
  import contextlib
16
15
  from collections.abc import AsyncIterator, Awaitable, Iterator
17
- from typing import Any, BinaryIO, TypeVar, cast
16
+ from typing import TYPE_CHECKING, Any, BinaryIO, TypeVar, cast
18
17
 
19
18
  from . import _core
20
19
  from .client import DatabricksClient
21
20
 
21
+ if TYPE_CHECKING:
22
+ import asyncio
23
+
22
24
  __all__ = [
23
25
  "HEARTBEAT",
24
26
  "QueryTimeout",
@@ -131,6 +133,10 @@ async def await_with_heartbeat(
131
133
  Yields HEARTBEAT zero or more times, then yields the awaitable's real
132
134
  result exactly once. Re-raises whatever `aw` raised, or QueryTimeout if
133
135
  `total_timeout_s` elapses first."""
136
+ # Imported here, not at module level: asyncio costs ~27 ms to import and
137
+ # only this coroutine (already running inside a loop) needs it.
138
+ import asyncio
139
+
134
140
  task: asyncio.Task[T] = asyncio.ensure_future(aw)
135
141
  loop = asyncio.get_running_loop()
136
142
  deadline = loop.time() + total_timeout_s if total_timeout_s is not None else None
@@ -52,6 +52,9 @@ def _table_to_rows(table: Any) -> list[Row]:
52
52
  if table.num_rows == 0:
53
53
  return []
54
54
  try:
55
+ from arro3.core import Table
56
+
57
+ table = Table.from_arrow(table)
55
58
  columns = [table.column(i).combine_chunks().to_pylist() for i in range(table.num_columns)]
56
59
  except ModuleNotFoundError as exc:
57
60
  raise ModuleNotFoundError(
File without changes