periplus-python-sdk 0.6.1__tar.gz → 0.8.0__tar.gz
This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
- periplus_python_sdk-0.8.0/PKG-INFO +70 -0
- periplus_python_sdk-0.8.0/README.md +52 -0
- {periplus_python_sdk-0.6.1 → periplus_python_sdk-0.8.0}/pyproject.toml +1 -1
- periplus_python_sdk-0.8.0/src/periplus_python_sdk.egg-info/PKG-INFO +70 -0
- {periplus_python_sdk-0.6.1 → periplus_python_sdk-0.8.0}/src/periplus_python_sdk.egg-info/SOURCES.txt +3 -1
- {periplus_python_sdk-0.6.1 → periplus_python_sdk-0.8.0}/src/periplus_sdk/client.py +37 -16
- {periplus_python_sdk-0.6.1 → periplus_python_sdk-0.8.0}/src/periplus_sdk/dbapi.py +98 -77
- periplus_python_sdk-0.8.0/src/periplus_sdk/sql_api.py +47 -0
- {periplus_python_sdk-0.6.1 → periplus_python_sdk-0.8.0}/src/periplus_sdk/sqlalchemy.py +35 -28
- periplus_python_sdk-0.8.0/src/periplus_sdk/stream.py +134 -0
- {periplus_python_sdk-0.6.1 → periplus_python_sdk-0.8.0}/src/periplus_sdk/types.py +7 -2
- {periplus_python_sdk-0.6.1 → periplus_python_sdk-0.8.0}/tests/test_client.py +19 -24
- {periplus_python_sdk-0.6.1 → periplus_python_sdk-0.8.0}/tests/test_dbapi.py +20 -16
- periplus_python_sdk-0.8.0/tests/test_notebook.py +143 -0
- {periplus_python_sdk-0.6.1 → periplus_python_sdk-0.8.0}/tests/test_sql_api.py +7 -6
- periplus_python_sdk-0.8.0/tests/test_stream.py +79 -0
- periplus_python_sdk-0.6.1/PKG-INFO +0 -234
- periplus_python_sdk-0.6.1/README.md +0 -216
- periplus_python_sdk-0.6.1/src/periplus_python_sdk.egg-info/PKG-INFO +0 -234
- periplus_python_sdk-0.6.1/src/periplus_sdk/sql_api.py +0 -25
- periplus_python_sdk-0.6.1/tests/test_notebook.py +0 -88
- {periplus_python_sdk-0.6.1 → periplus_python_sdk-0.8.0}/LICENSE +0 -0
- {periplus_python_sdk-0.6.1 → periplus_python_sdk-0.8.0}/NOTICE +0 -0
- {periplus_python_sdk-0.6.1 → periplus_python_sdk-0.8.0}/setup.cfg +0 -0
- {periplus_python_sdk-0.6.1 → periplus_python_sdk-0.8.0}/src/periplus_python_sdk.egg-info/dependency_links.txt +0 -0
- {periplus_python_sdk-0.6.1 → periplus_python_sdk-0.8.0}/src/periplus_python_sdk.egg-info/entry_points.txt +0 -0
- {periplus_python_sdk-0.6.1 → periplus_python_sdk-0.8.0}/src/periplus_python_sdk.egg-info/requires.txt +0 -0
- {periplus_python_sdk-0.6.1 → periplus_python_sdk-0.8.0}/src/periplus_python_sdk.egg-info/top_level.txt +0 -0
- {periplus_python_sdk-0.6.1 → periplus_python_sdk-0.8.0}/src/periplus_sdk/__init__.py +0 -0
- {periplus_python_sdk-0.6.1 → periplus_python_sdk-0.8.0}/src/periplus_sdk/errors.py +0 -0
- {periplus_python_sdk-0.6.1 → periplus_python_sdk-0.8.0}/src/periplus_sdk/py.typed +0 -0
|
@@ -0,0 +1,70 @@
|
|
|
1
|
+
Metadata-Version: 2.4
|
|
2
|
+
Name: periplus-python-sdk
|
|
3
|
+
Version: 0.8.0
|
|
4
|
+
Summary: Read-only Python client for the public Periplus query API
|
|
5
|
+
License-Expression: Apache-2.0
|
|
6
|
+
Project-URL: Repository, https://github.com/elei-io/periplus
|
|
7
|
+
Project-URL: Issues, https://github.com/elei-io/periplus/issues
|
|
8
|
+
Requires-Python: >=3.11
|
|
9
|
+
Description-Content-Type: text/markdown
|
|
10
|
+
License-File: LICENSE
|
|
11
|
+
License-File: NOTICE
|
|
12
|
+
Requires-Dist: httpx>=0.28
|
|
13
|
+
Requires-Dist: pydantic<3,>=2.12
|
|
14
|
+
Requires-Dist: sqlalchemy<3,>=2.0
|
|
15
|
+
Provides-Extra: notebook
|
|
16
|
+
Requires-Dist: marimo[sql]>=0.24.1; extra == "notebook"
|
|
17
|
+
Dynamic: license-file
|
|
18
|
+
|
|
19
|
+
# Periplus Python SDK
|
|
20
|
+
|
|
21
|
+
Read-only access to the public ClickHouse corpus through the Periplus HTTP API.
|
|
22
|
+
Install the SDK from the same checkout as your deployment:
|
|
23
|
+
|
|
24
|
+
```sh
|
|
25
|
+
python -m pip install ./packages/periplus-python-sdk
|
|
26
|
+
```
|
|
27
|
+
|
|
28
|
+
```python
|
|
29
|
+
from periplus_sdk import Client
|
|
30
|
+
|
|
31
|
+
with Client("http://localhost:8080") as client:
|
|
32
|
+
result = client.execute(
|
|
33
|
+
"SELECT capture_id, url FROM public_v1.captures LIMIT ?", [10]
|
|
34
|
+
)
|
|
35
|
+
print(result.columns, result.rows)
|
|
36
|
+
```
|
|
37
|
+
|
|
38
|
+
Use `PERIPLUS_PUBLIC_URL` to omit the URL argument. No database credentials or
|
|
39
|
+
service token are needed for the public gateway. `AsyncClient` provides async
|
|
40
|
+
methods. `prepare` validates/explains SQL; `execute` returns typed columns, rows,
|
|
41
|
+
truncation and query metadata; `helpers` describes the installed public views.
|
|
42
|
+
|
|
43
|
+
The public schema is `public_v1`. HTML joins use `document_id` plus node index;
|
|
44
|
+
`document_id` identifies retained bytes and their HTML interpretation. Raw-byte hashes stay internal.
|
|
45
|
+
There is one query endpoint, with no experimental fallback. Client errors preserve
|
|
46
|
+
server categories and do not automatically retry executed queries.
|
|
47
|
+
|
|
48
|
+
For notebook/SQLAlchemy integration:
|
|
49
|
+
|
|
50
|
+
```python
|
|
51
|
+
from periplus_sdk import sql_api
|
|
52
|
+
from sqlalchemy import text
|
|
53
|
+
|
|
54
|
+
engine = sql_api.create_engine(base_url="http://localhost:8080")
|
|
55
|
+
with engine.connect() as connection:
|
|
56
|
+
print(connection.execute(text("SELECT url FROM public_v1.captures LIMIT 5")).all())
|
|
57
|
+
engine.dispose()
|
|
58
|
+
```
|
|
59
|
+
|
|
60
|
+
Marimo discovers the five public views through the helper catalogue. Column
|
|
61
|
+
reflection uses `SELECT * ... LIMIT 0` to retrieve native types without reading
|
|
62
|
+
corpus rows; no `SHOW`, `DESCRIBE`, or system-table access is required.
|
|
63
|
+
|
|
64
|
+
The DB-API connection advertises the ClickHouse dialect and converts native
|
|
65
|
+
nullable integer, decimal, date and datetime types. Nested types retain JSON wire
|
|
66
|
+
values. It is read-only: there are no client transactions or writable sessions.
|
|
67
|
+
Streaming cursors expose incomplete/truncated results explicitly; configure
|
|
68
|
+
`allow_partial` only when partial results suit the application.
|
|
69
|
+
|
|
70
|
+
See [the public schema](../../docs/SCHEMA.md) and [query boundary](../../docs/QUERY.md).
|
|
@@ -0,0 +1,52 @@
|
|
|
1
|
+
# Periplus Python SDK
|
|
2
|
+
|
|
3
|
+
Read-only access to the public ClickHouse corpus through the Periplus HTTP API.
|
|
4
|
+
Install the SDK from the same checkout as your deployment:
|
|
5
|
+
|
|
6
|
+
```sh
|
|
7
|
+
python -m pip install ./packages/periplus-python-sdk
|
|
8
|
+
```
|
|
9
|
+
|
|
10
|
+
```python
|
|
11
|
+
from periplus_sdk import Client
|
|
12
|
+
|
|
13
|
+
with Client("http://localhost:8080") as client:
|
|
14
|
+
result = client.execute(
|
|
15
|
+
"SELECT capture_id, url FROM public_v1.captures LIMIT ?", [10]
|
|
16
|
+
)
|
|
17
|
+
print(result.columns, result.rows)
|
|
18
|
+
```
|
|
19
|
+
|
|
20
|
+
Use `PERIPLUS_PUBLIC_URL` to omit the URL argument. No database credentials or
|
|
21
|
+
service token are needed for the public gateway. `AsyncClient` provides async
|
|
22
|
+
methods. `prepare` validates/explains SQL; `execute` returns typed columns, rows,
|
|
23
|
+
truncation and query metadata; `helpers` describes the installed public views.
|
|
24
|
+
|
|
25
|
+
The public schema is `public_v1`. HTML joins use `document_id` plus node index;
|
|
26
|
+
`document_id` identifies retained bytes and their HTML interpretation. Raw-byte hashes stay internal.
|
|
27
|
+
There is one query endpoint, with no experimental fallback. Client errors preserve
|
|
28
|
+
server categories and do not automatically retry executed queries.
|
|
29
|
+
|
|
30
|
+
For notebook/SQLAlchemy integration:
|
|
31
|
+
|
|
32
|
+
```python
|
|
33
|
+
from periplus_sdk import sql_api
|
|
34
|
+
from sqlalchemy import text
|
|
35
|
+
|
|
36
|
+
engine = sql_api.create_engine(base_url="http://localhost:8080")
|
|
37
|
+
with engine.connect() as connection:
|
|
38
|
+
print(connection.execute(text("SELECT url FROM public_v1.captures LIMIT 5")).all())
|
|
39
|
+
engine.dispose()
|
|
40
|
+
```
|
|
41
|
+
|
|
42
|
+
Marimo discovers the five public views through the helper catalogue. Column
|
|
43
|
+
reflection uses `SELECT * ... LIMIT 0` to retrieve native types without reading
|
|
44
|
+
corpus rows; no `SHOW`, `DESCRIBE`, or system-table access is required.
|
|
45
|
+
|
|
46
|
+
The DB-API connection advertises the ClickHouse dialect and converts native
|
|
47
|
+
nullable integer, decimal, date and datetime types. Nested types retain JSON wire
|
|
48
|
+
values. It is read-only: there are no client transactions or writable sessions.
|
|
49
|
+
Streaming cursors expose incomplete/truncated results explicitly; configure
|
|
50
|
+
`allow_partial` only when partial results suit the application.
|
|
51
|
+
|
|
52
|
+
See [the public schema](../../docs/SCHEMA.md) and [query boundary](../../docs/QUERY.md).
|
|
@@ -0,0 +1,70 @@
|
|
|
1
|
+
Metadata-Version: 2.4
|
|
2
|
+
Name: periplus-python-sdk
|
|
3
|
+
Version: 0.8.0
|
|
4
|
+
Summary: Read-only Python client for the public Periplus query API
|
|
5
|
+
License-Expression: Apache-2.0
|
|
6
|
+
Project-URL: Repository, https://github.com/elei-io/periplus
|
|
7
|
+
Project-URL: Issues, https://github.com/elei-io/periplus/issues
|
|
8
|
+
Requires-Python: >=3.11
|
|
9
|
+
Description-Content-Type: text/markdown
|
|
10
|
+
License-File: LICENSE
|
|
11
|
+
License-File: NOTICE
|
|
12
|
+
Requires-Dist: httpx>=0.28
|
|
13
|
+
Requires-Dist: pydantic<3,>=2.12
|
|
14
|
+
Requires-Dist: sqlalchemy<3,>=2.0
|
|
15
|
+
Provides-Extra: notebook
|
|
16
|
+
Requires-Dist: marimo[sql]>=0.24.1; extra == "notebook"
|
|
17
|
+
Dynamic: license-file
|
|
18
|
+
|
|
19
|
+
# Periplus Python SDK
|
|
20
|
+
|
|
21
|
+
Read-only access to the public ClickHouse corpus through the Periplus HTTP API.
|
|
22
|
+
Install the SDK from the same checkout as your deployment:
|
|
23
|
+
|
|
24
|
+
```sh
|
|
25
|
+
python -m pip install ./packages/periplus-python-sdk
|
|
26
|
+
```
|
|
27
|
+
|
|
28
|
+
```python
|
|
29
|
+
from periplus_sdk import Client
|
|
30
|
+
|
|
31
|
+
with Client("http://localhost:8080") as client:
|
|
32
|
+
result = client.execute(
|
|
33
|
+
"SELECT capture_id, url FROM public_v1.captures LIMIT ?", [10]
|
|
34
|
+
)
|
|
35
|
+
print(result.columns, result.rows)
|
|
36
|
+
```
|
|
37
|
+
|
|
38
|
+
Use `PERIPLUS_PUBLIC_URL` to omit the URL argument. No database credentials or
|
|
39
|
+
service token are needed for the public gateway. `AsyncClient` provides async
|
|
40
|
+
methods. `prepare` validates/explains SQL; `execute` returns typed columns, rows,
|
|
41
|
+
truncation and query metadata; `helpers` describes the installed public views.
|
|
42
|
+
|
|
43
|
+
The public schema is `public_v1`. HTML joins use `document_id` plus node index;
|
|
44
|
+
`document_id` identifies retained bytes and their HTML interpretation. Raw-byte hashes stay internal.
|
|
45
|
+
There is one query endpoint, with no experimental fallback. Client errors preserve
|
|
46
|
+
server categories and do not automatically retry executed queries.
|
|
47
|
+
|
|
48
|
+
For notebook/SQLAlchemy integration:
|
|
49
|
+
|
|
50
|
+
```python
|
|
51
|
+
from periplus_sdk import sql_api
|
|
52
|
+
from sqlalchemy import text
|
|
53
|
+
|
|
54
|
+
engine = sql_api.create_engine(base_url="http://localhost:8080")
|
|
55
|
+
with engine.connect() as connection:
|
|
56
|
+
print(connection.execute(text("SELECT url FROM public_v1.captures LIMIT 5")).all())
|
|
57
|
+
engine.dispose()
|
|
58
|
+
```
|
|
59
|
+
|
|
60
|
+
Marimo discovers the five public views through the helper catalogue. Column
|
|
61
|
+
reflection uses `SELECT * ... LIMIT 0` to retrieve native types without reading
|
|
62
|
+
corpus rows; no `SHOW`, `DESCRIBE`, or system-table access is required.
|
|
63
|
+
|
|
64
|
+
The DB-API connection advertises the ClickHouse dialect and converts native
|
|
65
|
+
nullable integer, decimal, date and datetime types. Nested types retain JSON wire
|
|
66
|
+
values. It is read-only: there are no client transactions or writable sessions.
|
|
67
|
+
Streaming cursors expose incomplete/truncated results explicitly; configure
|
|
68
|
+
`allow_partial` only when partial results suit the application.
|
|
69
|
+
|
|
70
|
+
See [the public schema](../../docs/SCHEMA.md) and [query boundary](../../docs/QUERY.md).
|
{periplus_python_sdk-0.6.1 → periplus_python_sdk-0.8.0}/src/periplus_python_sdk.egg-info/SOURCES.txt
RENAMED
|
@@ -15,8 +15,10 @@ src/periplus_sdk/errors.py
|
|
|
15
15
|
src/periplus_sdk/py.typed
|
|
16
16
|
src/periplus_sdk/sql_api.py
|
|
17
17
|
src/periplus_sdk/sqlalchemy.py
|
|
18
|
+
src/periplus_sdk/stream.py
|
|
18
19
|
src/periplus_sdk/types.py
|
|
19
20
|
tests/test_client.py
|
|
20
21
|
tests/test_dbapi.py
|
|
21
22
|
tests/test_notebook.py
|
|
22
|
-
tests/test_sql_api.py
|
|
23
|
+
tests/test_sql_api.py
|
|
24
|
+
tests/test_stream.py
|
|
@@ -79,10 +79,11 @@ def _payload(sql: str, parameters: Sequence[JsonValue] | None, schema_version: s
|
|
|
79
79
|
class Client:
|
|
80
80
|
"""Reusable synchronous public query client. Close it or use a with block."""
|
|
81
81
|
|
|
82
|
-
def __init__(self, base_url: str | None = None, *, timeout: float =
|
|
83
|
-
if mode not in {"stable"
|
|
84
|
-
raise ConfigurationError("
|
|
85
|
-
self.
|
|
82
|
+
def __init__(self, base_url: str | None = None, *, timeout: float = 620, mode: Literal["stable"] = "stable"):
|
|
83
|
+
if mode not in {"stable"}:
|
|
84
|
+
raise ConfigurationError("Only the stable public catalogue is available.")
|
|
85
|
+
self.schema_version = "public_v1"
|
|
86
|
+
self._query_path = "api/query/"
|
|
86
87
|
self._http = httpx.Client(**_options(base_url, timeout))
|
|
87
88
|
|
|
88
89
|
def __enter__(self) -> Client:
|
|
@@ -101,11 +102,30 @@ class Client:
|
|
|
101
102
|
raise TransportError("Could not complete the public query request.") from None
|
|
102
103
|
return _decode(response, model)
|
|
103
104
|
|
|
104
|
-
def prepare(self, sql: str, parameters: Sequence[JsonValue] | None = None, *, schema_version: str =
|
|
105
|
-
return self._request("POST", "prep", PreparedQuery, json=_payload(sql, parameters, schema_version))
|
|
105
|
+
def prepare(self, sql: str, parameters: Sequence[JsonValue] | None = None, *, schema_version: str | None = None) -> PreparedQuery:
|
|
106
|
+
return self._request("POST", "prep", PreparedQuery, json=_payload(sql, parameters, schema_version if schema_version is not None else self.schema_version))
|
|
106
107
|
|
|
107
|
-
def execute(self, sql: str, parameters: Sequence[JsonValue] | None = None, *, schema_version: str =
|
|
108
|
-
return self._request("POST", "exec", QueryResult, json=_payload(sql, parameters, schema_version))
|
|
108
|
+
def execute(self, sql: str, parameters: Sequence[JsonValue] | None = None, *, schema_version: str | None = None) -> QueryResult:
|
|
109
|
+
return self._request("POST", "exec", QueryResult, json=_payload(sql, parameters, schema_version if schema_version is not None else self.schema_version))
|
|
110
|
+
|
|
111
|
+
def stream(self, sql: str, parameters: Sequence[JsonValue] | None = None, *,
|
|
112
|
+
schema_version: str | None = None, allow_partial: bool = False):
|
|
113
|
+
"""Stream batches from one snapshot; use as a context manager for early exit."""
|
|
114
|
+
from .stream import MEDIA_TYPE, QueryStream
|
|
115
|
+
try:
|
|
116
|
+
request = self._http.build_request("POST", self._query_path + "exec",
|
|
117
|
+
headers={"accept": MEDIA_TYPE}, json=_payload(sql, parameters, schema_version if schema_version is not None else self.schema_version))
|
|
118
|
+
response = self._http.send(request, stream=True)
|
|
119
|
+
try:
|
|
120
|
+
if not response.is_success:
|
|
121
|
+
response.read()
|
|
122
|
+
_decode(response, QueryResult)
|
|
123
|
+
return QueryStream(response, allow_partial=allow_partial)
|
|
124
|
+
except BaseException:
|
|
125
|
+
response.close()
|
|
126
|
+
raise
|
|
127
|
+
except httpx.RequestError:
|
|
128
|
+
raise TransportError("Could not open the public query stream.") from None
|
|
109
129
|
|
|
110
130
|
def helpers(self) -> QueryHelpers:
|
|
111
131
|
return self._request("GET", "helpers", QueryHelpers)
|
|
@@ -114,10 +134,11 @@ class Client:
|
|
|
114
134
|
class AsyncClient:
|
|
115
135
|
"""Reusable asynchronous public query client. Use an async with block."""
|
|
116
136
|
|
|
117
|
-
def __init__(self, base_url: str | None = None, *, timeout: float =
|
|
118
|
-
if mode not in {"stable"
|
|
119
|
-
raise ConfigurationError("
|
|
120
|
-
self.
|
|
137
|
+
def __init__(self, base_url: str | None = None, *, timeout: float = 620, mode: Literal["stable"] = "stable"):
|
|
138
|
+
if mode not in {"stable"}:
|
|
139
|
+
raise ConfigurationError("Only the stable public catalogue is available.")
|
|
140
|
+
self.schema_version = "public_v1"
|
|
141
|
+
self._query_path = "api/query/"
|
|
121
142
|
self._http = httpx.AsyncClient(**_options(base_url, timeout))
|
|
122
143
|
|
|
123
144
|
async def __aenter__(self) -> AsyncClient:
|
|
@@ -136,11 +157,11 @@ class AsyncClient:
|
|
|
136
157
|
raise TransportError("Could not complete the public query request.") from None
|
|
137
158
|
return _decode(response, model)
|
|
138
159
|
|
|
139
|
-
async def prepare(self, sql: str, parameters: Sequence[JsonValue] | None = None, *, schema_version: str =
|
|
140
|
-
return await self._request("POST", "prep", PreparedQuery, json=_payload(sql, parameters, schema_version))
|
|
160
|
+
async def prepare(self, sql: str, parameters: Sequence[JsonValue] | None = None, *, schema_version: str | None = None) -> PreparedQuery:
|
|
161
|
+
return await self._request("POST", "prep", PreparedQuery, json=_payload(sql, parameters, schema_version if schema_version is not None else self.schema_version))
|
|
141
162
|
|
|
142
|
-
async def execute(self, sql: str, parameters: Sequence[JsonValue] | None = None, *, schema_version: str =
|
|
143
|
-
return await self._request("POST", "exec", QueryResult, json=_payload(sql, parameters, schema_version))
|
|
163
|
+
async def execute(self, sql: str, parameters: Sequence[JsonValue] | None = None, *, schema_version: str | None = None) -> QueryResult:
|
|
164
|
+
return await self._request("POST", "exec", QueryResult, json=_payload(sql, parameters, schema_version if schema_version is not None else self.schema_version))
|
|
144
165
|
|
|
145
166
|
async def helpers(self) -> QueryHelpers:
|
|
146
167
|
return await self._request("GET", "helpers", QueryHelpers)
|
|
@@ -1,7 +1,7 @@
|
|
|
1
1
|
"""Read-only DB-API 2.0 connection over the public query API.
|
|
2
2
|
|
|
3
|
-
Each execute is an independent server snapshot. Fetching consumes a bounded
|
|
4
|
-
|
|
3
|
+
Each execute is an independent server snapshot. Fetching consumes a bounded stream.
|
|
4
|
+
Connections and cursors must not be shared by threads.
|
|
5
5
|
"""
|
|
6
6
|
from __future__ import annotations
|
|
7
7
|
|
|
@@ -11,11 +11,10 @@ from collections.abc import Sequence
|
|
|
11
11
|
from datetime import date, datetime, time
|
|
12
12
|
from decimal import Decimal
|
|
13
13
|
from typing import Any, Literal
|
|
14
|
-
import warnings
|
|
15
14
|
|
|
16
15
|
from .client import Client
|
|
17
16
|
from .errors import ApiError, ConfigurationError, PeriplusError, ResponseError, TransportError
|
|
18
|
-
from .
|
|
17
|
+
from .stream import StreamResult
|
|
19
18
|
|
|
20
19
|
apilevel = "2.0"
|
|
21
20
|
threadsafety = 1
|
|
@@ -26,10 +25,6 @@ class Warning(builtins.Warning):
|
|
|
26
25
|
"""DB-API warning."""
|
|
27
26
|
|
|
28
27
|
|
|
29
|
-
class TruncationWarning(Warning):
|
|
30
|
-
"""The server returned only part of the query result."""
|
|
31
|
-
|
|
32
|
-
|
|
33
28
|
class Error(PeriplusError):
|
|
34
29
|
"""Base DB-API error; API failures preserve their safe error attributes."""
|
|
35
30
|
|
|
@@ -91,54 +86,47 @@ def TimestampFromTicks(ticks: float) -> datetime:
|
|
|
91
86
|
return datetime.fromtimestamp(ticks)
|
|
92
87
|
|
|
93
88
|
|
|
94
|
-
_INTEGER_TYPES = {
|
|
95
|
-
|
|
96
|
-
|
|
97
|
-
|
|
98
|
-
|
|
99
|
-
|
|
89
|
+
_INTEGER_TYPES = {f'{prefix}Int{bits}' for prefix in ('', 'U') for bits in (8,16,32,64,128,256)}
|
|
90
|
+
_FLOAT_TYPES = {'Float32', 'Float64'}
|
|
91
|
+
|
|
92
|
+
|
|
93
|
+
def _base_type(value: str) -> str:
|
|
94
|
+
while value.startswith(('Nullable(', 'LowCardinality(')):
|
|
95
|
+
value = value[value.index('(')+1:-1]
|
|
96
|
+
return value
|
|
100
97
|
|
|
101
98
|
|
|
102
99
|
class _TypeCategory:
|
|
103
|
-
def __init__(self, names: set[str],
|
|
104
|
-
self.names, self.
|
|
100
|
+
def __init__(self, names: set[str], prefixes: tuple[str, ...] = ()):
|
|
101
|
+
self.names, self.prefixes = names, prefixes
|
|
105
102
|
|
|
106
103
|
def __eq__(self, other: object) -> bool:
|
|
107
|
-
|
|
104
|
+
if not isinstance(other, str):return False
|
|
105
|
+
value = _base_type(other)
|
|
106
|
+
return value in self.names or value.startswith(self.prefixes)
|
|
108
107
|
|
|
109
108
|
|
|
110
|
-
STRING = _TypeCategory({
|
|
111
|
-
BINARY = _TypeCategory(
|
|
112
|
-
NUMBER = _TypeCategory(_INTEGER_TYPES | _FLOAT_TYPES | {
|
|
113
|
-
DATETIME = _TypeCategory({
|
|
109
|
+
STRING = _TypeCategory({'String','UUID','JSON'}, ('FixedString(', 'Enum'))
|
|
110
|
+
BINARY = _TypeCategory(set())
|
|
111
|
+
NUMBER = _TypeCategory(_INTEGER_TYPES | _FLOAT_TYPES | {'Bool'}, ('Decimal',))
|
|
112
|
+
DATETIME = _TypeCategory({'Date','Date32'}, ('DateTime','Time'))
|
|
114
113
|
ROWID = _TypeCategory(set())
|
|
115
114
|
|
|
116
115
|
|
|
117
116
|
def _value(value: Any, sql_type: str) -> Any:
|
|
118
|
-
if value is None:
|
|
119
|
-
|
|
120
|
-
if sql_type in _INTEGER_TYPES:
|
|
121
|
-
|
|
122
|
-
if sql_type
|
|
123
|
-
|
|
124
|
-
if sql_type
|
|
125
|
-
return
|
|
126
|
-
|
|
127
|
-
if sql_type
|
|
128
|
-
try:
|
|
129
|
-
|
|
130
|
-
|
|
131
|
-
return value
|
|
132
|
-
if sql_type in _TIME_TYPES:
|
|
133
|
-
return time.fromisoformat(value)
|
|
134
|
-
if sql_type in _TIMESTAMP_TYPES:
|
|
135
|
-
try:
|
|
136
|
-
return datetime.fromisoformat(value)
|
|
137
|
-
except ValueError:
|
|
138
|
-
return value
|
|
139
|
-
if sql_type == "BLOB":
|
|
140
|
-
return base64.b64decode(value, validate=True)
|
|
141
|
-
# UUIDs remain strings, and nested/other types retain their JSON wire values.
|
|
117
|
+
if value is None:return None
|
|
118
|
+
sql_type = _base_type(sql_type)
|
|
119
|
+
if sql_type in _INTEGER_TYPES:return int(value)
|
|
120
|
+
if sql_type in _FLOAT_TYPES:return float(value)
|
|
121
|
+
if sql_type.startswith('Decimal'):return Decimal(str(value))
|
|
122
|
+
if sql_type == 'Bool':return value in (True, 1, '1', 'true')
|
|
123
|
+
if sql_type in ('Date','Date32'):
|
|
124
|
+
try:return date.fromisoformat(value)
|
|
125
|
+
except ValueError:return value
|
|
126
|
+
if sql_type.startswith('DateTime'):
|
|
127
|
+
try:return datetime.fromisoformat(value)
|
|
128
|
+
except ValueError:return value
|
|
129
|
+
if sql_type.startswith('Time'):return time.fromisoformat(value)
|
|
142
130
|
return value
|
|
143
131
|
|
|
144
132
|
|
|
@@ -147,24 +135,30 @@ def _parameter(value: Any) -> Any:
|
|
|
147
135
|
return value
|
|
148
136
|
if isinstance(value, (Decimal, date, time)):
|
|
149
137
|
return str(value) if isinstance(value, Decimal) else value.isoformat()
|
|
150
|
-
|
|
138
|
+
if isinstance(value, (list, tuple)):
|
|
139
|
+
if any(isinstance(item, (list, tuple, dict)) for item in value):
|
|
140
|
+
raise ProgrammingError("Collection parameters must be one-dimensional lists of scalar values.")
|
|
141
|
+
return [_parameter(item) for item in value]
|
|
142
|
+
raise ProgrammingError("Parameters must be scalar values or scalar lists; use explicit SQL casts for typed values.")
|
|
151
143
|
|
|
152
144
|
|
|
153
145
|
class Connection:
|
|
154
146
|
"""Marimo-discoverable, read-only connection; commit is a no-op."""
|
|
155
147
|
|
|
156
|
-
dialect = "
|
|
148
|
+
dialect = "clickhouse"
|
|
157
149
|
|
|
158
|
-
def __init__(self, base_url: str | None = None, *, timeout: float =
|
|
159
|
-
mode: Literal["stable"
|
|
160
|
-
schema_version: str =
|
|
150
|
+
def __init__(self, base_url: str | None = None, *, timeout: float = 620,
|
|
151
|
+
mode: Literal["stable"] = "stable",
|
|
152
|
+
schema_version: str | None = None, allow_partial: bool = False):
|
|
161
153
|
try:
|
|
162
154
|
self._client = Client(base_url, timeout=timeout, mode=mode)
|
|
163
155
|
except ConfigurationError as exc:
|
|
164
156
|
raise InterfaceError(str(exc)) from exc
|
|
165
|
-
self.schema_version = schema_version
|
|
157
|
+
self.schema_version = schema_version if schema_version is not None else ("public_v1")
|
|
158
|
+
self.allow_partial = allow_partial
|
|
159
|
+
self._cursors = set()
|
|
166
160
|
self.closed = False
|
|
167
|
-
self.last_result:
|
|
161
|
+
self.last_result: StreamResult | None = None
|
|
168
162
|
|
|
169
163
|
def _check(self) -> None:
|
|
170
164
|
if self.closed:
|
|
@@ -172,7 +166,9 @@ class Connection:
|
|
|
172
166
|
|
|
173
167
|
def cursor(self) -> Cursor:
|
|
174
168
|
self._check()
|
|
175
|
-
|
|
169
|
+
cursor = Cursor(self)
|
|
170
|
+
self._cursors.add(cursor)
|
|
171
|
+
return cursor
|
|
176
172
|
|
|
177
173
|
def execute(self, operation: str, parameters: Sequence[Any] | None = None) -> Cursor:
|
|
178
174
|
cursor = self.cursor()
|
|
@@ -192,6 +188,8 @@ class Connection:
|
|
|
192
188
|
|
|
193
189
|
def close(self) -> None:
|
|
194
190
|
if not self.closed:
|
|
191
|
+
for cursor in list(self._cursors):
|
|
192
|
+
cursor.close()
|
|
195
193
|
self._client.close()
|
|
196
194
|
self.closed = True
|
|
197
195
|
self.last_result = None
|
|
@@ -204,25 +202,26 @@ class Connection:
|
|
|
204
202
|
self.close()
|
|
205
203
|
|
|
206
204
|
|
|
207
|
-
def connect(base_url: str | None = None, *, timeout: float =
|
|
208
|
-
mode: Literal["stable"
|
|
209
|
-
schema_version: str =
|
|
210
|
-
return Connection(base_url, timeout=timeout, mode=mode, schema_version=schema_version)
|
|
205
|
+
def connect(base_url: str | None = None, *, timeout: float = 620,
|
|
206
|
+
mode: Literal["stable"] = "stable",
|
|
207
|
+
schema_version: str | None = None, allow_partial: bool = False) -> Connection:
|
|
208
|
+
return Connection(base_url, timeout=timeout, mode=mode, schema_version=schema_version, allow_partial=allow_partial)
|
|
211
209
|
|
|
212
210
|
|
|
213
211
|
class Cursor:
|
|
214
|
-
"""
|
|
212
|
+
"""Incremental cursor. Metadata stays available without retaining consumed rows."""
|
|
215
213
|
|
|
216
214
|
arraysize = 1
|
|
217
215
|
|
|
218
216
|
def __init__(self, connection: Connection):
|
|
219
217
|
self.connection = connection
|
|
220
218
|
self.closed = False
|
|
221
|
-
self.result:
|
|
219
|
+
self.result: StreamResult | None = None
|
|
222
220
|
self.description: list[tuple[Any, ...]] | None = None
|
|
223
221
|
self.rowcount = -1
|
|
224
222
|
self._rows: list[tuple[Any, ...]] = []
|
|
225
223
|
self._position = 0
|
|
224
|
+
self._stream = None
|
|
226
225
|
|
|
227
226
|
def _check(self, *, result: bool = False) -> None:
|
|
228
227
|
self.connection._check()
|
|
@@ -233,6 +232,9 @@ class Cursor:
|
|
|
233
232
|
|
|
234
233
|
def execute(self, operation: str, parameters: Sequence[Any] | None = None) -> Cursor:
|
|
235
234
|
self._check()
|
|
235
|
+
if self._stream is not None:
|
|
236
|
+
self._stream.close()
|
|
237
|
+
self._stream = None
|
|
236
238
|
self.result, self.description, self.rowcount = None, None, -1
|
|
237
239
|
self._rows, self._position = [], 0
|
|
238
240
|
self.connection.last_result = None
|
|
@@ -242,7 +244,9 @@ class Cursor:
|
|
|
242
244
|
raise ProgrammingError("Use a positional parameter sequence with ? placeholders.")
|
|
243
245
|
values = [_parameter(v) for v in parameters] if parameters is not None else []
|
|
244
246
|
try:
|
|
245
|
-
|
|
247
|
+
self._stream = self.connection._client.stream(operation, values, schema_version=self.connection.schema_version,
|
|
248
|
+
allow_partial=self.connection.allow_partial)
|
|
249
|
+
result = self._stream.result
|
|
246
250
|
except ApiError as exc:
|
|
247
251
|
error = ProgrammingError if exc.code == "sql_invalid" else OperationalError
|
|
248
252
|
raise error(str(exc), status_code=exc.status_code, code=exc.code,
|
|
@@ -251,26 +255,37 @@ class Cursor:
|
|
|
251
255
|
raise OperationalError(str(exc)) from exc
|
|
252
256
|
except ResponseError as exc:
|
|
253
257
|
raise InterfaceError(str(exc)) from exc
|
|
254
|
-
if len(result.columns) != len(result.types) or any(len(r) != len(result.columns) for r in result.rows):
|
|
255
|
-
raise InterfaceError("Query columns, types and rows have inconsistent widths.")
|
|
256
|
-
try:
|
|
257
|
-
rows = [tuple(_value(v, t) for v, t in zip(row, result.types, strict=True)) for row in result.rows]
|
|
258
|
-
except (ValueError, TypeError, ArithmeticError) as exc:
|
|
259
|
-
raise DataError("Query value does not match its SQL type.") from exc
|
|
260
258
|
self.result = self.connection.last_result = result
|
|
261
259
|
self.description = [(name, kind, None, None, None, None, None)
|
|
262
260
|
for name, kind in zip(result.columns, result.types, strict=True)]
|
|
263
|
-
self._rows = rows
|
|
264
|
-
self.rowcount = -1 if result.truncated else len(rows)
|
|
265
|
-
if result.truncated:
|
|
266
|
-
warnings.warn(f"Periplus returned a truncated result ({len(rows)} rows); inspect connection.last_result or cursor.result. Fetching does not retrieve additional rows.",
|
|
267
|
-
TruncationWarning, stacklevel=2)
|
|
268
261
|
return self
|
|
269
262
|
|
|
263
|
+
def _batch(self):
|
|
264
|
+
try:
|
|
265
|
+
batch = next(self._stream)
|
|
266
|
+
self._rows = [tuple(_value(v, t) for v, t in zip(row, self.result.types, strict=True)) for row in batch]
|
|
267
|
+
self._position = 0
|
|
268
|
+
return True
|
|
269
|
+
except StopIteration:
|
|
270
|
+
self.rowcount = -1 if self.result.truncated else self.result.row_count
|
|
271
|
+
self._rows, self._position = [], 0
|
|
272
|
+
return False
|
|
273
|
+
except ApiError as exc:
|
|
274
|
+
raise OperationalError(str(exc), status_code=exc.status_code, code=exc.code,
|
|
275
|
+
retry_after_seconds=exc.retry_after_seconds) from exc
|
|
276
|
+
except TransportError as exc:
|
|
277
|
+
raise OperationalError(str(exc)) from exc
|
|
278
|
+
except ResponseError as exc:
|
|
279
|
+
raise InterfaceError(str(exc)) from exc
|
|
280
|
+
except (ValueError, TypeError, ArithmeticError) as exc:
|
|
281
|
+
self._stream.close()
|
|
282
|
+
raise DataError("Query value does not match its SQL type.") from exc
|
|
283
|
+
|
|
270
284
|
def fetchone(self) -> tuple[Any, ...] | None:
|
|
271
285
|
self._check(result=True)
|
|
272
|
-
|
|
273
|
-
|
|
286
|
+
while self._position == len(self._rows):
|
|
287
|
+
if not self._batch():
|
|
288
|
+
return None
|
|
274
289
|
row = self._rows[self._position]
|
|
275
290
|
self._position += 1
|
|
276
291
|
return row
|
|
@@ -280,14 +295,17 @@ class Cursor:
|
|
|
280
295
|
size = self.arraysize if size is None else size
|
|
281
296
|
if not isinstance(size, int) or size < 0:
|
|
282
297
|
raise ProgrammingError("Fetch size must be a non-negative integer.")
|
|
283
|
-
|
|
284
|
-
|
|
285
|
-
|
|
298
|
+
rows = []
|
|
299
|
+
for _ in range(size):
|
|
300
|
+
row = self.fetchone()
|
|
301
|
+
if row is None:
|
|
302
|
+
break
|
|
303
|
+
rows.append(row)
|
|
286
304
|
return rows
|
|
287
305
|
|
|
288
306
|
def fetchall(self) -> list[tuple[Any, ...]]:
|
|
289
307
|
self._check(result=True)
|
|
290
|
-
return
|
|
308
|
+
return list(self)
|
|
291
309
|
|
|
292
310
|
def executemany(self, operation: str, seq_of_parameters: Any) -> None:
|
|
293
311
|
self._check()
|
|
@@ -300,6 +318,9 @@ class Cursor:
|
|
|
300
318
|
self._check()
|
|
301
319
|
|
|
302
320
|
def close(self) -> None:
|
|
321
|
+
if self._stream is not None:
|
|
322
|
+
self._stream.close()
|
|
323
|
+
self.connection._cursors.discard(self)
|
|
303
324
|
self.closed = True
|
|
304
325
|
self._rows = []
|
|
305
326
|
self.result = None
|
|
@@ -0,0 +1,47 @@
|
|
|
1
|
+
"""Notebook-friendly SQL engine over the public Periplus query API."""
|
|
2
|
+
from __future__ import annotations
|
|
3
|
+
|
|
4
|
+
from typing import Literal
|
|
5
|
+
|
|
6
|
+
from sqlalchemy import create_engine as _create_engine, event
|
|
7
|
+
from sqlalchemy.engine import Engine, URL
|
|
8
|
+
|
|
9
|
+
|
|
10
|
+
def create_engine(
|
|
11
|
+
base_url: str | None = None,
|
|
12
|
+
*,
|
|
13
|
+
mode: Literal["stable"] = "stable",
|
|
14
|
+
timeout: float = 620,
|
|
15
|
+
schema_version: str | None = None,
|
|
16
|
+
allow_partial: bool = False,
|
|
17
|
+
) -> Engine:
|
|
18
|
+
"""Create a SQLAlchemy engine recognized by marimo and other SQL tools.
|
|
19
|
+
|
|
20
|
+
The public URL defaults to PERIPLUS_PUBLIC_URL. Connections are opened lazily;
|
|
21
|
+
dispose the engine when finished. Each query uses an independent server snapshot.
|
|
22
|
+
"""
|
|
23
|
+
engine = _create_engine(
|
|
24
|
+
URL.create("periplus", database="periplus"),
|
|
25
|
+
connect_args={"base_url": base_url, "mode": mode, "timeout": timeout, "schema_version": schema_version, "allow_partial": allow_partial},
|
|
26
|
+
)
|
|
27
|
+
|
|
28
|
+
event.listen(engine, "before_execute", _parameters, retval=True)
|
|
29
|
+
return engine
|
|
30
|
+
|
|
31
|
+
|
|
32
|
+
def _parameters(connection, clauseelement, multiparams, params, execution_options):
|
|
33
|
+
bindings = execution_options.get("periplus_parameters", {})
|
|
34
|
+
if bindings and not multiparams:
|
|
35
|
+
names = clauseelement.compile().params
|
|
36
|
+
params = {**{name: value for name, value in bindings.items() if name in names}, **params}
|
|
37
|
+
return clauseelement, multiparams, params
|
|
38
|
+
|
|
39
|
+
|
|
40
|
+
def bind(engine: Engine, **parameters) -> Engine:
|
|
41
|
+
"""Create a marimo-discoverable engine with named SQL parameters for this cell.
|
|
42
|
+
|
|
43
|
+
Uses the original engine's pool. No query, upload, or server state is created.
|
|
44
|
+
Write :name placeholders in SQL cells; lists can be CAST(:ids AS VARCHAR[]).
|
|
45
|
+
"""
|
|
46
|
+
from .dbapi import _parameter
|
|
47
|
+
return engine.execution_options(periplus_parameters={name: _parameter(value) for name, value in parameters.items()})
|