periplus-python-sdk 0.6.0__tar.gz → 0.7.0__tar.gz

This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
Files changed (27) hide show
  1. {periplus_python_sdk-0.6.0/src/periplus_python_sdk.egg-info → periplus_python_sdk-0.7.0}/PKG-INFO +84 -17
  2. {periplus_python_sdk-0.6.0 → periplus_python_sdk-0.7.0}/README.md +83 -16
  3. {periplus_python_sdk-0.6.0 → periplus_python_sdk-0.7.0}/pyproject.toml +1 -1
  4. {periplus_python_sdk-0.6.0 → periplus_python_sdk-0.7.0/src/periplus_python_sdk.egg-info}/PKG-INFO +84 -17
  5. {periplus_python_sdk-0.6.0 → periplus_python_sdk-0.7.0}/src/periplus_python_sdk.egg-info/SOURCES.txt +3 -1
  6. {periplus_python_sdk-0.6.0 → periplus_python_sdk-0.7.0}/src/periplus_sdk/client.py +21 -2
  7. {periplus_python_sdk-0.6.0 → periplus_python_sdk-0.7.0}/src/periplus_sdk/dbapi.py +64 -36
  8. periplus_python_sdk-0.7.0/src/periplus_sdk/sql_api.py +47 -0
  9. {periplus_python_sdk-0.6.0 → periplus_python_sdk-0.7.0}/src/periplus_sdk/sqlalchemy.py +6 -6
  10. periplus_python_sdk-0.7.0/src/periplus_sdk/stream.py +134 -0
  11. {periplus_python_sdk-0.6.0 → periplus_python_sdk-0.7.0}/src/periplus_sdk/types.py +3 -0
  12. {periplus_python_sdk-0.6.0 → periplus_python_sdk-0.7.0}/tests/test_client.py +13 -1
  13. {periplus_python_sdk-0.6.0 → periplus_python_sdk-0.7.0}/tests/test_dbapi.py +13 -9
  14. {periplus_python_sdk-0.6.0 → periplus_python_sdk-0.7.0}/tests/test_notebook.py +41 -6
  15. {periplus_python_sdk-0.6.0 → periplus_python_sdk-0.7.0}/tests/test_sql_api.py +3 -2
  16. periplus_python_sdk-0.7.0/tests/test_stream.py +79 -0
  17. periplus_python_sdk-0.6.0/src/periplus_sdk/sql_api.py +0 -25
  18. {periplus_python_sdk-0.6.0 → periplus_python_sdk-0.7.0}/LICENSE +0 -0
  19. {periplus_python_sdk-0.6.0 → periplus_python_sdk-0.7.0}/NOTICE +0 -0
  20. {periplus_python_sdk-0.6.0 → periplus_python_sdk-0.7.0}/setup.cfg +0 -0
  21. {periplus_python_sdk-0.6.0 → periplus_python_sdk-0.7.0}/src/periplus_python_sdk.egg-info/dependency_links.txt +0 -0
  22. {periplus_python_sdk-0.6.0 → periplus_python_sdk-0.7.0}/src/periplus_python_sdk.egg-info/entry_points.txt +0 -0
  23. {periplus_python_sdk-0.6.0 → periplus_python_sdk-0.7.0}/src/periplus_python_sdk.egg-info/requires.txt +0 -0
  24. {periplus_python_sdk-0.6.0 → periplus_python_sdk-0.7.0}/src/periplus_python_sdk.egg-info/top_level.txt +0 -0
  25. {periplus_python_sdk-0.6.0 → periplus_python_sdk-0.7.0}/src/periplus_sdk/__init__.py +0 -0
  26. {periplus_python_sdk-0.6.0 → periplus_python_sdk-0.7.0}/src/periplus_sdk/errors.py +0 -0
  27. {periplus_python_sdk-0.6.0 → periplus_python_sdk-0.7.0}/src/periplus_sdk/py.typed +0 -0
@@ -1,6 +1,6 @@
1
1
  Metadata-Version: 2.4
2
2
  Name: periplus-python-sdk
3
- Version: 0.6.0
3
+ Version: 0.7.0
4
4
  Summary: Read-only Python client for the public Periplus query API
5
5
  License-Expression: Apache-2.0
6
6
  Project-URL: Repository, https://github.com/elei-io/periplus
@@ -43,7 +43,7 @@ The client reuses HTTP connections; close it with a context manager or `close()`
43
43
  Install the notebook integration from PyPI:
44
44
 
45
45
  ```sh
46
- uv add "periplus-python-sdk[notebook]>=0.6.0"
46
+ uv add "periplus-python-sdk[notebook]>=0.7.0"
47
47
  ```
48
48
 
49
49
  In a Python setup cell, create a SQLAlchemy engine:
@@ -62,7 +62,7 @@ FROM public_v1.capture
62
62
  LIMIT 10
63
63
  ```
64
64
 
65
- Marimo displays the result as a table. Expand **pp → public_v1** in Data Sources
65
+ Marimo displays the result as a table. Expand **pp → periplus → public_v1** in Data Sources
66
66
  to discover views and expand a view to load its columns for SQL completion.
67
67
  Discovery uses bounded `SHOW TABLES` and `DESCRIBE` through the same public API;
68
68
  no internal catalogue or storage credentials are used. Truncated discovery fails
@@ -82,7 +82,7 @@ captures = mo.sql(
82
82
  ```
83
83
 
84
84
  Set `mode="experimental"` for the experimental service. Omit the URL to use
85
- `PERIPLUS_PUBLIC_URL`. Optional `timeout=140` and `schema_version="public_v1"`
85
+ `PERIPLUS_PUBLIC_URL`. Optional `timeout=620` and `schema_version="public_v1"`
86
86
  arguments configure the client deadline and public schema. Run `pp.dispose()` when finished. This is a read-only
87
87
  SQLAlchemy dialect for textual SQL and reflection, not a writable ORM backend.
88
88
  Each statement has its own server snapshot; SQLAlchemy transaction blocks do not
@@ -90,7 +90,7 @@ provide a shared snapshot or rollback. The adapter makes no transaction requests
90
90
 
91
91
  A complete notebook is in `examples/notebook.py`. The integration is tested with
92
92
  marimo 0.24.1 and SQLAlchemy 2.x. SQLAlchemy is included in the standard SDK install; the `notebook` extra adds
93
- marimo. Existing marimo environments only need `uv add "periplus-python-sdk>=0.6.0"`.
93
+ marimo. Existing marimo environments only need `uv add "periplus-python-sdk>=0.7.0"`.
94
94
  The returned object is a standard SQLAlchemy Engine, also usable with pandas and
95
95
  ordinary Python scripts. Engine creation is lazy; the first query opens a connection.
96
96
 
@@ -112,14 +112,17 @@ with connect("https://periplus.dev", mode="stable") as connection:
112
112
  Connections expose `cursor`, `execute`, `close`, and context managers. Cursors
113
113
  support `execute`, `fetchone`, `fetchmany`, `fetchall`, iteration, and close.
114
114
  Use positional `?` parameters. Decimal and temporal parameters are sent as
115
- strings; use explicit SQL casts. Binary and nested parameters are not supported
116
- by this adapter. Fetching only consumes the bounded result already received;
117
- it never issues pagination or retries. Connections/cursors are not thread-shared.
115
+ strings; use explicit SQL casts. One-dimensional lists/tuples of these scalar values are supported; use an explicit array cast
116
+ for empty or typed lists. Binary, mappings, and nested collection parameters are unsupported. Fetching consumes incremental batches from one HTTP response and one source snapshot;
117
+ it never issues pagination or retries. Close a cursor early to stop delivery. Connections/cursors are not thread-shared.
118
118
  `commit()` is a no-op; `rollback()` and `executemany()` are unsupported.
119
119
 
120
- `cursor.result` preserves the original query response. `connection.last_result`
120
+ `cursor.result` preserves stream metadata and completion information without retaining consumed rows. `connection.last_result`
121
121
  also retains it after marimo closes a cursor; a new execution clears it first.
122
- Truncation emits `periplus_sdk.dbapi.TruncationWarning` and sets `rowcount` to -1.
122
+ Truncation raises `OperationalError` with `code="result_limit"` when fetching reaches the terminal frame.
123
+ `rowcount` stays -1 until completion, then reports the delivered count. `result.complete` means the
124
+ terminal frame arrived; inspect `result.truncated` separately when opting into partial data.
125
+ Use `allow_partial=True` on the engine or connection only when incomplete results are intentional.
123
126
  DB-API failures use the standard exception hierarchy in `periplus_sdk.dbapi`;
124
127
  HTTP errors retain `status_code`, `code`, and `retry_after_seconds`.
125
128
 
@@ -185,7 +188,7 @@ Use `aclose()` when managing an async client's lifetime explicitly.
185
188
  20-second server deadline. Always inspect `truncated`. The SDK does not silently fetch more rows or retry.
186
189
  - `ApiError` exposes `status_code`, safe `code`, and `retry_after_seconds` when supplied.
187
190
  `TransportError` means HTTP failed; `ResponseError` means a malformed successful response.
188
- The client timeout defaults to 140 seconds and can be set with `timeout=`. A timeout or local
191
+ The client timeout defaults to 620 seconds and can be set with `timeout=`. A timeout or local
189
192
  cancellation does not guarantee server cancellation. Redirects are not followed automatically.
190
193
  - Preparation and execution are attributed to `sdk` in the existing private query history.
191
194
  Original SQL and parameters are retained for 30 days; result rows are not stored. Recording is
@@ -196,10 +199,10 @@ Use `aclose()` when managing an async client's lifetime explicitly.
196
199
  Install the public-v1 client from PyPI:
197
200
 
198
201
  ```sh
199
- python -m pip install "periplus-python-sdk>=0.6.0"
202
+ python -m pip install "periplus-python-sdk>=0.7.0"
200
203
  ```
201
204
 
202
- Version 0.6.0 supports the current public-v1 contract. For production, configure
205
+ Version 0.6.1 supports the current public-v1 contract. For production, configure
203
206
  `PERIPLUS_PUBLIC_URL=https://periplus.dev`; no API token is required.
204
207
  Run the installed package against an available public app:
205
208
 
@@ -211,11 +214,11 @@ PERIPLUS_PUBLIC_URL=http://localhost:8080 python packages/periplus-python-sdk/ex
211
214
 
212
215
  Repository CI publishes immutable releases from tags named
213
216
  `periplus-python-sdk-v<version>`. The tag must exactly match the static version
214
- in `pyproject.toml`; for example, version `0.6.0` is released with:
217
+ in `pyproject.toml`; for example, version `0.7.0` is released with:
215
218
 
216
219
  ```sh
217
- git tag periplus-python-sdk-v0.6.0
218
- git push origin periplus-python-sdk-v0.6.0
220
+ git tag periplus-python-sdk-v0.7.0
221
+ git push origin periplus-python-sdk-v0.7.0
219
222
  ```
220
223
 
221
224
  PyPI publishing uses Trusted Publishing rather than a stored API token. The
@@ -225,10 +228,74 @@ that GitHub environment with required reviewers before the first release.
225
228
 
226
229
  ## Public v1
227
230
 
228
- Install the updated SDK from PyPI with `python -m pip install "periplus-python-sdk>=0.6.0"`. The previously published 0.2.0 release predates this contract. `prepare` and `execute` accept keyword-only `schema_version="public_v1"` (the default); responses preserve `schema_version` separately from `source_snapshot`. Unavailable versions are rejected by the server.
231
+ Install the updated SDK from PyPI with `python -m pip install "periplus-python-sdk>=0.7.0"`. The previously published 0.2.0 release predates this contract. `prepare` and `execute` accept keyword-only `schema_version="public_v1"` (the default); responses preserve `schema_version` separately from `source_snapshot`. Unavailable versions are rejected by the server.
229
232
 
230
233
  ## License
231
234
 
232
235
  Copyright (c) 2026 Ekku Leivonen (elei.io). Licensed under [Apache-2.0](LICENSE);
233
236
  see [NOTICE](NOTICE). The server and other repository packages have separate
234
237
  licensing described in the root LICENSING.md.
238
+
239
+ ## Python filtering back into marimo SQL
240
+
241
+ The existing engine streams by default. For SQL → Python filtering → SQL, bind the
242
+ filtered IDs to another engine variable; it shares the original connection pool and
243
+ creates no server-side table or upload. Marimo discovers it like any other SQLAlchemy engine.
244
+
245
+ ```python
246
+ pp = sql_api.create_engine("https://periplus.dev")
247
+
248
+ # In a Python cell, after filtering your initial SQL dataframe:
249
+ selected = sql_api.bind(pp, content_ids=qualified_pages["content_id"].to_list())
250
+
251
+ # In the next SQL cell (engine: selected):
252
+ raw_elements = mo.sql(
253
+ """
254
+ SELECT content_id, node_index, parent_index, tag,
255
+ trim(text_direct) AS text, attributes['class'] AS elem_class
256
+ FROM public_v1.html_element
257
+ WHERE content_id IN (SELECT unnest(CAST(:content_ids AS VARCHAR[])))
258
+ AND text_direct IS NOT NULL AND trim(text_direct) <> ''
259
+ """,
260
+ engine=selected,
261
+ )
262
+ ```
263
+
264
+ Keep SQL strings plain: `:content_ids` is a bound parameter, not an f-string.
265
+ An empty list selects no elements; duplicate IDs do not multiply rows. Re-running the
266
+ binding cell snapshots the new Python list into its engine options. Bindings only apply
267
+ to matching SQL placeholders, so catalogue discovery remains available. Explicit
268
+ SQLAlchemy execution parameters take precedence over engine bindings.
269
+
270
+ Marimo collects the stream into a local dataframe, which must fit notebook memory.
271
+ Its dataframe conversion currently infers types and may lose empty/all-null column
272
+ information; the DB-API cursor always exposes server column names and SQL types.
273
+ Initial SQL results and subsequent element results both reject truncation by default.
274
+ Each separate SQL execution has its own snapshot; Python filtering does not pin the
275
+ first query's snapshot across the round trip.
276
+
277
+ ## Consume batches without a dataframe
278
+
279
+ ```python
280
+ with Client("https://periplus.dev") as client:
281
+ with client.stream("SELECT content_id FROM public_v1.prose") as stream:
282
+ for rows in stream:
283
+ process_batch(rows)
284
+ print(stream.result.row_count, stream.result.source_snapshot)
285
+ print(stream.result.limits)
286
+ ```
287
+
288
+ Batches retain JSON wire values with SQL types in `stream.result.types`. DB-API fetching
289
+ performs scalar Python decoding. The synchronous `Client.stream()` raises on truncation
290
+ unless `allow_partial=True`; missing completion, malformed frames, timeouts, and broken
291
+ connections always fail. Do not treat already-consumed batches as a complete dataset
292
+ until iteration finishes successfully. Use context managers when stopping early.
293
+
294
+ Admin Query limits controls requests (default 4 MiB, ceiling 16 MiB), parameter values
295
+ (default 100,000, ceiling 1,000,000), results (ceiling 10 million rows and 1 GiB), and
296
+ execution duration (ceiling 600 seconds). Parameter counting includes collection containers
297
+ and their values. Existing result/duration defaults remain 1,000 rows, 8 MiB, and 20 seconds.
298
+ Operators must raise these settings for larger workloads. No SDK parameter grants a larger
299
+ server budget. Stream metadata includes effective limits; input rejections name the budget.
300
+ Deploy the API, query service, public gateway and SDK together after applying Alembic
301
+ revision `20260911_0016`. Ingress must permit the configured request size and stream duration.
@@ -25,7 +25,7 @@ The client reuses HTTP connections; close it with a context manager or `close()`
25
25
  Install the notebook integration from PyPI:
26
26
 
27
27
  ```sh
28
- uv add "periplus-python-sdk[notebook]>=0.6.0"
28
+ uv add "periplus-python-sdk[notebook]>=0.7.0"
29
29
  ```
30
30
 
31
31
  In a Python setup cell, create a SQLAlchemy engine:
@@ -44,7 +44,7 @@ FROM public_v1.capture
44
44
  LIMIT 10
45
45
  ```
46
46
 
47
- Marimo displays the result as a table. Expand **pp → public_v1** in Data Sources
47
+ Marimo displays the result as a table. Expand **pp → periplus → public_v1** in Data Sources
48
48
  to discover views and expand a view to load its columns for SQL completion.
49
49
  Discovery uses bounded `SHOW TABLES` and `DESCRIBE` through the same public API;
50
50
  no internal catalogue or storage credentials are used. Truncated discovery fails
@@ -64,7 +64,7 @@ captures = mo.sql(
64
64
  ```
65
65
 
66
66
  Set `mode="experimental"` for the experimental service. Omit the URL to use
67
- `PERIPLUS_PUBLIC_URL`. Optional `timeout=140` and `schema_version="public_v1"`
67
+ `PERIPLUS_PUBLIC_URL`. Optional `timeout=620` and `schema_version="public_v1"`
68
68
  arguments configure the client deadline and public schema. Run `pp.dispose()` when finished. This is a read-only
69
69
  SQLAlchemy dialect for textual SQL and reflection, not a writable ORM backend.
70
70
  Each statement has its own server snapshot; SQLAlchemy transaction blocks do not
@@ -72,7 +72,7 @@ provide a shared snapshot or rollback. The adapter makes no transaction requests
72
72
 
73
73
  A complete notebook is in `examples/notebook.py`. The integration is tested with
74
74
  marimo 0.24.1 and SQLAlchemy 2.x. SQLAlchemy is included in the standard SDK install; the `notebook` extra adds
75
- marimo. Existing marimo environments only need `uv add "periplus-python-sdk>=0.6.0"`.
75
+ marimo. Existing marimo environments only need `uv add "periplus-python-sdk>=0.7.0"`.
76
76
  The returned object is a standard SQLAlchemy Engine, also usable with pandas and
77
77
  ordinary Python scripts. Engine creation is lazy; the first query opens a connection.
78
78
 
@@ -94,14 +94,17 @@ with connect("https://periplus.dev", mode="stable") as connection:
94
94
  Connections expose `cursor`, `execute`, `close`, and context managers. Cursors
95
95
  support `execute`, `fetchone`, `fetchmany`, `fetchall`, iteration, and close.
96
96
  Use positional `?` parameters. Decimal and temporal parameters are sent as
97
- strings; use explicit SQL casts. Binary and nested parameters are not supported
98
- by this adapter. Fetching only consumes the bounded result already received;
99
- it never issues pagination or retries. Connections/cursors are not thread-shared.
97
+ strings; use explicit SQL casts. One-dimensional lists/tuples of these scalar values are supported; use an explicit array cast
98
+ for empty or typed lists. Binary, mappings, and nested collection parameters are unsupported. Fetching consumes incremental batches from one HTTP response and one source snapshot;
99
+ it never issues pagination or retries. Close a cursor early to stop delivery. Connections/cursors are not thread-shared.
100
100
  `commit()` is a no-op; `rollback()` and `executemany()` are unsupported.
101
101
 
102
- `cursor.result` preserves the original query response. `connection.last_result`
102
+ `cursor.result` preserves stream metadata and completion information without retaining consumed rows. `connection.last_result`
103
103
  also retains it after marimo closes a cursor; a new execution clears it first.
104
- Truncation emits `periplus_sdk.dbapi.TruncationWarning` and sets `rowcount` to -1.
104
+ Truncation raises `OperationalError` with `code="result_limit"` when fetching reaches the terminal frame.
105
+ `rowcount` stays -1 until completion, then reports the delivered count. `result.complete` means the
106
+ terminal frame arrived; inspect `result.truncated` separately when opting into partial data.
107
+ Use `allow_partial=True` on the engine or connection only when incomplete results are intentional.
105
108
  DB-API failures use the standard exception hierarchy in `periplus_sdk.dbapi`;
106
109
  HTTP errors retain `status_code`, `code`, and `retry_after_seconds`.
107
110
 
@@ -167,7 +170,7 @@ Use `aclose()` when managing an async client's lifetime explicitly.
167
170
  20-second server deadline. Always inspect `truncated`. The SDK does not silently fetch more rows or retry.
168
171
  - `ApiError` exposes `status_code`, safe `code`, and `retry_after_seconds` when supplied.
169
172
  `TransportError` means HTTP failed; `ResponseError` means a malformed successful response.
170
- The client timeout defaults to 140 seconds and can be set with `timeout=`. A timeout or local
173
+ The client timeout defaults to 620 seconds and can be set with `timeout=`. A timeout or local
171
174
  cancellation does not guarantee server cancellation. Redirects are not followed automatically.
172
175
  - Preparation and execution are attributed to `sdk` in the existing private query history.
173
176
  Original SQL and parameters are retained for 30 days; result rows are not stored. Recording is
@@ -178,10 +181,10 @@ Use `aclose()` when managing an async client's lifetime explicitly.
178
181
  Install the public-v1 client from PyPI:
179
182
 
180
183
  ```sh
181
- python -m pip install "periplus-python-sdk>=0.6.0"
184
+ python -m pip install "periplus-python-sdk>=0.7.0"
182
185
  ```
183
186
 
184
- Version 0.6.0 supports the current public-v1 contract. For production, configure
187
+ Version 0.6.1 supports the current public-v1 contract. For production, configure
185
188
  `PERIPLUS_PUBLIC_URL=https://periplus.dev`; no API token is required.
186
189
  Run the installed package against an available public app:
187
190
 
@@ -193,11 +196,11 @@ PERIPLUS_PUBLIC_URL=http://localhost:8080 python packages/periplus-python-sdk/ex
193
196
 
194
197
  Repository CI publishes immutable releases from tags named
195
198
  `periplus-python-sdk-v<version>`. The tag must exactly match the static version
196
- in `pyproject.toml`; for example, version `0.6.0` is released with:
199
+ in `pyproject.toml`; for example, version `0.7.0` is released with:
197
200
 
198
201
  ```sh
199
- git tag periplus-python-sdk-v0.6.0
200
- git push origin periplus-python-sdk-v0.6.0
202
+ git tag periplus-python-sdk-v0.7.0
203
+ git push origin periplus-python-sdk-v0.7.0
201
204
  ```
202
205
 
203
206
  PyPI publishing uses Trusted Publishing rather than a stored API token. The
@@ -207,10 +210,74 @@ that GitHub environment with required reviewers before the first release.
207
210
 
208
211
  ## Public v1
209
212
 
210
- Install the updated SDK from PyPI with `python -m pip install "periplus-python-sdk>=0.6.0"`. The previously published 0.2.0 release predates this contract. `prepare` and `execute` accept keyword-only `schema_version="public_v1"` (the default); responses preserve `schema_version` separately from `source_snapshot`. Unavailable versions are rejected by the server.
213
+ Install the updated SDK from PyPI with `python -m pip install "periplus-python-sdk>=0.7.0"`. The previously published 0.2.0 release predates this contract. `prepare` and `execute` accept keyword-only `schema_version="public_v1"` (the default); responses preserve `schema_version` separately from `source_snapshot`. Unavailable versions are rejected by the server.
211
214
 
212
215
  ## License
213
216
 
214
217
  Copyright (c) 2026 Ekku Leivonen (elei.io). Licensed under [Apache-2.0](LICENSE);
215
218
  see [NOTICE](NOTICE). The server and other repository packages have separate
216
219
  licensing described in the root LICENSING.md.
220
+
221
+ ## Python filtering back into marimo SQL
222
+
223
+ The existing engine streams by default. For SQL → Python filtering → SQL, bind the
224
+ filtered IDs to another engine variable; it shares the original connection pool and
225
+ creates no server-side table or upload. Marimo discovers it like any other SQLAlchemy engine.
226
+
227
+ ```python
228
+ pp = sql_api.create_engine("https://periplus.dev")
229
+
230
+ # In a Python cell, after filtering your initial SQL dataframe:
231
+ selected = sql_api.bind(pp, content_ids=qualified_pages["content_id"].to_list())
232
+
233
+ # In the next SQL cell (engine: selected):
234
+ raw_elements = mo.sql(
235
+ """
236
+ SELECT content_id, node_index, parent_index, tag,
237
+ trim(text_direct) AS text, attributes['class'] AS elem_class
238
+ FROM public_v1.html_element
239
+ WHERE content_id IN (SELECT unnest(CAST(:content_ids AS VARCHAR[])))
240
+ AND text_direct IS NOT NULL AND trim(text_direct) <> ''
241
+ """,
242
+ engine=selected,
243
+ )
244
+ ```
245
+
246
+ Keep SQL strings plain: `:content_ids` is a bound parameter, not an f-string.
247
+ An empty list selects no elements; duplicate IDs do not multiply rows. Re-running the
248
+ binding cell snapshots the new Python list into its engine options. Bindings only apply
249
+ to matching SQL placeholders, so catalogue discovery remains available. Explicit
250
+ SQLAlchemy execution parameters take precedence over engine bindings.
251
+
252
+ Marimo collects the stream into a local dataframe, which must fit notebook memory.
253
+ Its dataframe conversion currently infers types and may lose empty/all-null column
254
+ information; the DB-API cursor always exposes server column names and SQL types.
255
+ Initial SQL results and subsequent element results both reject truncation by default.
256
+ Each separate SQL execution has its own snapshot; Python filtering does not pin the
257
+ first query's snapshot across the round trip.
258
+
259
+ ## Consume batches without a dataframe
260
+
261
+ ```python
262
+ with Client("https://periplus.dev") as client:
263
+ with client.stream("SELECT content_id FROM public_v1.prose") as stream:
264
+ for rows in stream:
265
+ process_batch(rows)
266
+ print(stream.result.row_count, stream.result.source_snapshot)
267
+ print(stream.result.limits)
268
+ ```
269
+
270
+ Batches retain JSON wire values with SQL types in `stream.result.types`. DB-API fetching
271
+ performs scalar Python decoding. The synchronous `Client.stream()` raises on truncation
272
+ unless `allow_partial=True`; missing completion, malformed frames, timeouts, and broken
273
+ connections always fail. Do not treat already-consumed batches as a complete dataset
274
+ until iteration finishes successfully. Use context managers when stopping early.
275
+
276
+ Admin Query limits controls requests (default 4 MiB, ceiling 16 MiB), parameter values
277
+ (default 100,000, ceiling 1,000,000), results (ceiling 10 million rows and 1 GiB), and
278
+ execution duration (ceiling 600 seconds). Parameter counting includes collection containers
279
+ and their values. Existing result/duration defaults remain 1,000 rows, 8 MiB, and 20 seconds.
280
+ Operators must raise these settings for larger workloads. No SDK parameter grants a larger
281
+ server budget. Stream metadata includes effective limits; input rejections name the budget.
282
+ Deploy the API, query service, public gateway and SDK together after applying Alembic
283
+ revision `20260911_0016`. Ingress must permit the configured request size and stream duration.
@@ -2,7 +2,7 @@
2
2
  license = "Apache-2.0"
3
3
  license-files = ["LICENSE", "NOTICE"]
4
4
  name = "periplus-python-sdk"
5
- version = "0.6.0"
5
+ version = "0.7.0"
6
6
  description = "Read-only Python client for the public Periplus query API"
7
7
  readme = "README.md"
8
8
  requires-python = ">=3.11"
@@ -1,6 +1,6 @@
1
1
  Metadata-Version: 2.4
2
2
  Name: periplus-python-sdk
3
- Version: 0.6.0
3
+ Version: 0.7.0
4
4
  Summary: Read-only Python client for the public Periplus query API
5
5
  License-Expression: Apache-2.0
6
6
  Project-URL: Repository, https://github.com/elei-io/periplus
@@ -43,7 +43,7 @@ The client reuses HTTP connections; close it with a context manager or `close()`
43
43
  Install the notebook integration from PyPI:
44
44
 
45
45
  ```sh
46
- uv add "periplus-python-sdk[notebook]>=0.6.0"
46
+ uv add "periplus-python-sdk[notebook]>=0.7.0"
47
47
  ```
48
48
 
49
49
  In a Python setup cell, create a SQLAlchemy engine:
@@ -62,7 +62,7 @@ FROM public_v1.capture
62
62
  LIMIT 10
63
63
  ```
64
64
 
65
- Marimo displays the result as a table. Expand **pp → public_v1** in Data Sources
65
+ Marimo displays the result as a table. Expand **pp → periplus → public_v1** in Data Sources
66
66
  to discover views and expand a view to load its columns for SQL completion.
67
67
  Discovery uses bounded `SHOW TABLES` and `DESCRIBE` through the same public API;
68
68
  no internal catalogue or storage credentials are used. Truncated discovery fails
@@ -82,7 +82,7 @@ captures = mo.sql(
82
82
  ```
83
83
 
84
84
  Set `mode="experimental"` for the experimental service. Omit the URL to use
85
- `PERIPLUS_PUBLIC_URL`. Optional `timeout=140` and `schema_version="public_v1"`
85
+ `PERIPLUS_PUBLIC_URL`. Optional `timeout=620` and `schema_version="public_v1"`
86
86
  arguments configure the client deadline and public schema. Run `pp.dispose()` when finished. This is a read-only
87
87
  SQLAlchemy dialect for textual SQL and reflection, not a writable ORM backend.
88
88
  Each statement has its own server snapshot; SQLAlchemy transaction blocks do not
@@ -90,7 +90,7 @@ provide a shared snapshot or rollback. The adapter makes no transaction requests
90
90
 
91
91
  A complete notebook is in `examples/notebook.py`. The integration is tested with
92
92
  marimo 0.24.1 and SQLAlchemy 2.x. SQLAlchemy is included in the standard SDK install; the `notebook` extra adds
93
- marimo. Existing marimo environments only need `uv add "periplus-python-sdk>=0.6.0"`.
93
+ marimo. Existing marimo environments only need `uv add "periplus-python-sdk>=0.7.0"`.
94
94
  The returned object is a standard SQLAlchemy Engine, also usable with pandas and
95
95
  ordinary Python scripts. Engine creation is lazy; the first query opens a connection.
96
96
 
@@ -112,14 +112,17 @@ with connect("https://periplus.dev", mode="stable") as connection:
112
112
  Connections expose `cursor`, `execute`, `close`, and context managers. Cursors
113
113
  support `execute`, `fetchone`, `fetchmany`, `fetchall`, iteration, and close.
114
114
  Use positional `?` parameters. Decimal and temporal parameters are sent as
115
- strings; use explicit SQL casts. Binary and nested parameters are not supported
116
- by this adapter. Fetching only consumes the bounded result already received;
117
- it never issues pagination or retries. Connections/cursors are not thread-shared.
115
+ strings; use explicit SQL casts. One-dimensional lists/tuples of these scalar values are supported; use an explicit array cast
116
+ for empty or typed lists. Binary, mappings, and nested collection parameters are unsupported. Fetching consumes incremental batches from one HTTP response and one source snapshot;
117
+ it never issues pagination or retries. Close a cursor early to stop delivery. Connections/cursors are not thread-shared.
118
118
  `commit()` is a no-op; `rollback()` and `executemany()` are unsupported.
119
119
 
120
- `cursor.result` preserves the original query response. `connection.last_result`
120
+ `cursor.result` preserves stream metadata and completion information without retaining consumed rows. `connection.last_result`
121
121
  also retains it after marimo closes a cursor; a new execution clears it first.
122
- Truncation emits `periplus_sdk.dbapi.TruncationWarning` and sets `rowcount` to -1.
122
+ Truncation raises `OperationalError` with `code="result_limit"` when fetching reaches the terminal frame.
123
+ `rowcount` stays -1 until completion, then reports the delivered count. `result.complete` means the
124
+ terminal frame arrived; inspect `result.truncated` separately when opting into partial data.
125
+ Use `allow_partial=True` on the engine or connection only when incomplete results are intentional.
123
126
  DB-API failures use the standard exception hierarchy in `periplus_sdk.dbapi`;
124
127
  HTTP errors retain `status_code`, `code`, and `retry_after_seconds`.
125
128
 
@@ -185,7 +188,7 @@ Use `aclose()` when managing an async client's lifetime explicitly.
185
188
  20-second server deadline. Always inspect `truncated`. The SDK does not silently fetch more rows or retry.
186
189
  - `ApiError` exposes `status_code`, safe `code`, and `retry_after_seconds` when supplied.
187
190
  `TransportError` means HTTP failed; `ResponseError` means a malformed successful response.
188
- The client timeout defaults to 140 seconds and can be set with `timeout=`. A timeout or local
191
+ The client timeout defaults to 620 seconds and can be set with `timeout=`. A timeout or local
189
192
  cancellation does not guarantee server cancellation. Redirects are not followed automatically.
190
193
  - Preparation and execution are attributed to `sdk` in the existing private query history.
191
194
  Original SQL and parameters are retained for 30 days; result rows are not stored. Recording is
@@ -196,10 +199,10 @@ Use `aclose()` when managing an async client's lifetime explicitly.
196
199
  Install the public-v1 client from PyPI:
197
200
 
198
201
  ```sh
199
- python -m pip install "periplus-python-sdk>=0.6.0"
202
+ python -m pip install "periplus-python-sdk>=0.7.0"
200
203
  ```
201
204
 
202
- Version 0.6.0 supports the current public-v1 contract. For production, configure
205
+ Version 0.6.1 supports the current public-v1 contract. For production, configure
203
206
  `PERIPLUS_PUBLIC_URL=https://periplus.dev`; no API token is required.
204
207
  Run the installed package against an available public app:
205
208
 
@@ -211,11 +214,11 @@ PERIPLUS_PUBLIC_URL=http://localhost:8080 python packages/periplus-python-sdk/ex
211
214
 
212
215
  Repository CI publishes immutable releases from tags named
213
216
  `periplus-python-sdk-v<version>`. The tag must exactly match the static version
214
- in `pyproject.toml`; for example, version `0.6.0` is released with:
217
+ in `pyproject.toml`; for example, version `0.7.0` is released with:
215
218
 
216
219
  ```sh
217
- git tag periplus-python-sdk-v0.6.0
218
- git push origin periplus-python-sdk-v0.6.0
220
+ git tag periplus-python-sdk-v0.7.0
221
+ git push origin periplus-python-sdk-v0.7.0
219
222
  ```
220
223
 
221
224
  PyPI publishing uses Trusted Publishing rather than a stored API token. The
@@ -225,10 +228,74 @@ that GitHub environment with required reviewers before the first release.
225
228
 
226
229
  ## Public v1
227
230
 
228
- Install the updated SDK from PyPI with `python -m pip install "periplus-python-sdk>=0.6.0"`. The previously published 0.2.0 release predates this contract. `prepare` and `execute` accept keyword-only `schema_version="public_v1"` (the default); responses preserve `schema_version` separately from `source_snapshot`. Unavailable versions are rejected by the server.
231
+ Install the updated SDK from PyPI with `python -m pip install "periplus-python-sdk>=0.7.0"`. The previously published 0.2.0 release predates this contract. `prepare` and `execute` accept keyword-only `schema_version="public_v1"` (the default); responses preserve `schema_version` separately from `source_snapshot`. Unavailable versions are rejected by the server.
229
232
 
230
233
  ## License
231
234
 
232
235
  Copyright (c) 2026 Ekku Leivonen (elei.io). Licensed under [Apache-2.0](LICENSE);
233
236
  see [NOTICE](NOTICE). The server and other repository packages have separate
234
237
  licensing described in the root LICENSING.md.
238
+
239
+ ## Python filtering back into marimo SQL
240
+
241
+ The existing engine streams by default. For SQL → Python filtering → SQL, bind the
242
+ filtered IDs to another engine variable; it shares the original connection pool and
243
+ creates no server-side table or upload. Marimo discovers it like any other SQLAlchemy engine.
244
+
245
+ ```python
246
+ pp = sql_api.create_engine("https://periplus.dev")
247
+
248
+ # In a Python cell, after filtering your initial SQL dataframe:
249
+ selected = sql_api.bind(pp, content_ids=qualified_pages["content_id"].to_list())
250
+
251
+ # In the next SQL cell (engine: selected):
252
+ raw_elements = mo.sql(
253
+ """
254
+ SELECT content_id, node_index, parent_index, tag,
255
+ trim(text_direct) AS text, attributes['class'] AS elem_class
256
+ FROM public_v1.html_element
257
+ WHERE content_id IN (SELECT unnest(CAST(:content_ids AS VARCHAR[])))
258
+ AND text_direct IS NOT NULL AND trim(text_direct) <> ''
259
+ """,
260
+ engine=selected,
261
+ )
262
+ ```
263
+
264
+ Keep SQL strings plain: `:content_ids` is a bound parameter, not an f-string.
265
+ An empty list selects no elements; duplicate IDs do not multiply rows. Re-running the
266
+ binding cell snapshots the new Python list into its engine options. Bindings only apply
267
+ to matching SQL placeholders, so catalogue discovery remains available. Explicit
268
+ SQLAlchemy execution parameters take precedence over engine bindings.
269
+
270
+ Marimo collects the stream into a local dataframe, which must fit notebook memory.
271
+ Its dataframe conversion currently infers types and may lose empty/all-null column
272
+ information; the DB-API cursor always exposes server column names and SQL types.
273
+ Initial SQL results and subsequent element results both reject truncation by default.
274
+ Each separate SQL execution has its own snapshot; Python filtering does not pin the
275
+ first query's snapshot across the round trip.
276
+
277
+ ## Consume batches without a dataframe
278
+
279
+ ```python
280
+ with Client("https://periplus.dev") as client:
281
+ with client.stream("SELECT content_id FROM public_v1.prose") as stream:
282
+ for rows in stream:
283
+ process_batch(rows)
284
+ print(stream.result.row_count, stream.result.source_snapshot)
285
+ print(stream.result.limits)
286
+ ```
287
+
288
+ Batches retain JSON wire values with SQL types in `stream.result.types`. DB-API fetching
289
+ performs scalar Python decoding. The synchronous `Client.stream()` raises on truncation
290
+ unless `allow_partial=True`; missing completion, malformed frames, timeouts, and broken
291
+ connections always fail. Do not treat already-consumed batches as a complete dataset
292
+ until iteration finishes successfully. Use context managers when stopping early.
293
+
294
+ Admin Query limits controls requests (default 4 MiB, ceiling 16 MiB), parameter values
295
+ (default 100,000, ceiling 1,000,000), results (ceiling 10 million rows and 1 GiB), and
296
+ execution duration (ceiling 600 seconds). Parameter counting includes collection containers
297
+ and their values. Existing result/duration defaults remain 1,000 rows, 8 MiB, and 20 seconds.
298
+ Operators must raise these settings for larger workloads. No SDK parameter grants a larger
299
+ server budget. Stream metadata includes effective limits; input rejections name the budget.
300
+ Deploy the API, query service, public gateway and SDK together after applying Alembic
301
+ revision `20260911_0016`. Ingress must permit the configured request size and stream duration.
@@ -15,8 +15,10 @@ src/periplus_sdk/errors.py
15
15
  src/periplus_sdk/py.typed
16
16
  src/periplus_sdk/sql_api.py
17
17
  src/periplus_sdk/sqlalchemy.py
18
+ src/periplus_sdk/stream.py
18
19
  src/periplus_sdk/types.py
19
20
  tests/test_client.py
20
21
  tests/test_dbapi.py
21
22
  tests/test_notebook.py
22
- tests/test_sql_api.py
23
+ tests/test_sql_api.py
24
+ tests/test_stream.py
@@ -79,7 +79,7 @@ def _payload(sql: str, parameters: Sequence[JsonValue] | None, schema_version: s
79
79
  class Client:
80
80
  """Reusable synchronous public query client. Close it or use a with block."""
81
81
 
82
- def __init__(self, base_url: str | None = None, *, timeout: float = 140, mode: Literal["stable", "experimental"] = "stable"):
82
+ def __init__(self, base_url: str | None = None, *, timeout: float = 620, mode: Literal["stable", "experimental"] = "stable"):
83
83
  if mode not in {"stable", "experimental"}:
84
84
  raise ConfigurationError("mode must be stable or experimental.")
85
85
  self._query_path = "api/query/experimental/" if mode == "experimental" else "api/query/"
@@ -107,6 +107,25 @@ class Client:
107
107
  def execute(self, sql: str, parameters: Sequence[JsonValue] | None = None, *, schema_version: str = "public_v1") -> QueryResult:
108
108
  return self._request("POST", "exec", QueryResult, json=_payload(sql, parameters, schema_version))
109
109
 
110
+ def stream(self, sql: str, parameters: Sequence[JsonValue] | None = None, *,
111
+ schema_version: str = "public_v1", allow_partial: bool = False):
112
+ """Stream batches from one snapshot; use as a context manager for early exit."""
113
+ from .stream import MEDIA_TYPE, QueryStream
114
+ try:
115
+ request = self._http.build_request("POST", self._query_path + "exec",
116
+ headers={"accept": MEDIA_TYPE}, json=_payload(sql, parameters, schema_version))
117
+ response = self._http.send(request, stream=True)
118
+ try:
119
+ if not response.is_success:
120
+ response.read()
121
+ _decode(response, QueryResult)
122
+ return QueryStream(response, allow_partial=allow_partial)
123
+ except BaseException:
124
+ response.close()
125
+ raise
126
+ except httpx.RequestError:
127
+ raise TransportError("Could not open the public query stream.") from None
128
+
110
129
  def helpers(self) -> QueryHelpers:
111
130
  return self._request("GET", "helpers", QueryHelpers)
112
131
 
@@ -114,7 +133,7 @@ class Client:
114
133
  class AsyncClient:
115
134
  """Reusable asynchronous public query client. Use an async with block."""
116
135
 
117
- def __init__(self, base_url: str | None = None, *, timeout: float = 140, mode: Literal["stable", "experimental"] = "stable"):
136
+ def __init__(self, base_url: str | None = None, *, timeout: float = 620, mode: Literal["stable", "experimental"] = "stable"):
118
137
  if mode not in {"stable", "experimental"}:
119
138
  raise ConfigurationError("mode must be stable or experimental.")
120
139
  self._query_path = "api/query/experimental/" if mode == "experimental" else "api/query/"