periplus-python-sdk 0.6.1__tar.gz → 0.8.0__tar.gz

This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
Files changed (31) hide show
  1. periplus_python_sdk-0.8.0/PKG-INFO +70 -0
  2. periplus_python_sdk-0.8.0/README.md +52 -0
  3. {periplus_python_sdk-0.6.1 → periplus_python_sdk-0.8.0}/pyproject.toml +1 -1
  4. periplus_python_sdk-0.8.0/src/periplus_python_sdk.egg-info/PKG-INFO +70 -0
  5. {periplus_python_sdk-0.6.1 → periplus_python_sdk-0.8.0}/src/periplus_python_sdk.egg-info/SOURCES.txt +3 -1
  6. {periplus_python_sdk-0.6.1 → periplus_python_sdk-0.8.0}/src/periplus_sdk/client.py +37 -16
  7. {periplus_python_sdk-0.6.1 → periplus_python_sdk-0.8.0}/src/periplus_sdk/dbapi.py +98 -77
  8. periplus_python_sdk-0.8.0/src/periplus_sdk/sql_api.py +47 -0
  9. {periplus_python_sdk-0.6.1 → periplus_python_sdk-0.8.0}/src/periplus_sdk/sqlalchemy.py +35 -28
  10. periplus_python_sdk-0.8.0/src/periplus_sdk/stream.py +134 -0
  11. {periplus_python_sdk-0.6.1 → periplus_python_sdk-0.8.0}/src/periplus_sdk/types.py +7 -2
  12. {periplus_python_sdk-0.6.1 → periplus_python_sdk-0.8.0}/tests/test_client.py +19 -24
  13. {periplus_python_sdk-0.6.1 → periplus_python_sdk-0.8.0}/tests/test_dbapi.py +20 -16
  14. periplus_python_sdk-0.8.0/tests/test_notebook.py +143 -0
  15. {periplus_python_sdk-0.6.1 → periplus_python_sdk-0.8.0}/tests/test_sql_api.py +7 -6
  16. periplus_python_sdk-0.8.0/tests/test_stream.py +79 -0
  17. periplus_python_sdk-0.6.1/PKG-INFO +0 -234
  18. periplus_python_sdk-0.6.1/README.md +0 -216
  19. periplus_python_sdk-0.6.1/src/periplus_python_sdk.egg-info/PKG-INFO +0 -234
  20. periplus_python_sdk-0.6.1/src/periplus_sdk/sql_api.py +0 -25
  21. periplus_python_sdk-0.6.1/tests/test_notebook.py +0 -88
  22. {periplus_python_sdk-0.6.1 → periplus_python_sdk-0.8.0}/LICENSE +0 -0
  23. {periplus_python_sdk-0.6.1 → periplus_python_sdk-0.8.0}/NOTICE +0 -0
  24. {periplus_python_sdk-0.6.1 → periplus_python_sdk-0.8.0}/setup.cfg +0 -0
  25. {periplus_python_sdk-0.6.1 → periplus_python_sdk-0.8.0}/src/periplus_python_sdk.egg-info/dependency_links.txt +0 -0
  26. {periplus_python_sdk-0.6.1 → periplus_python_sdk-0.8.0}/src/periplus_python_sdk.egg-info/entry_points.txt +0 -0
  27. {periplus_python_sdk-0.6.1 → periplus_python_sdk-0.8.0}/src/periplus_python_sdk.egg-info/requires.txt +0 -0
  28. {periplus_python_sdk-0.6.1 → periplus_python_sdk-0.8.0}/src/periplus_python_sdk.egg-info/top_level.txt +0 -0
  29. {periplus_python_sdk-0.6.1 → periplus_python_sdk-0.8.0}/src/periplus_sdk/__init__.py +0 -0
  30. {periplus_python_sdk-0.6.1 → periplus_python_sdk-0.8.0}/src/periplus_sdk/errors.py +0 -0
  31. {periplus_python_sdk-0.6.1 → periplus_python_sdk-0.8.0}/src/periplus_sdk/py.typed +0 -0
@@ -0,0 +1,70 @@
1
+ Metadata-Version: 2.4
2
+ Name: periplus-python-sdk
3
+ Version: 0.8.0
4
+ Summary: Read-only Python client for the public Periplus query API
5
+ License-Expression: Apache-2.0
6
+ Project-URL: Repository, https://github.com/elei-io/periplus
7
+ Project-URL: Issues, https://github.com/elei-io/periplus/issues
8
+ Requires-Python: >=3.11
9
+ Description-Content-Type: text/markdown
10
+ License-File: LICENSE
11
+ License-File: NOTICE
12
+ Requires-Dist: httpx>=0.28
13
+ Requires-Dist: pydantic<3,>=2.12
14
+ Requires-Dist: sqlalchemy<3,>=2.0
15
+ Provides-Extra: notebook
16
+ Requires-Dist: marimo[sql]>=0.24.1; extra == "notebook"
17
+ Dynamic: license-file
18
+
19
+ # Periplus Python SDK
20
+
21
+ Read-only access to the public ClickHouse corpus through the Periplus HTTP API.
22
+ Install the SDK from the same checkout as your deployment:
23
+
24
+ ```sh
25
+ python -m pip install ./packages/periplus-python-sdk
26
+ ```
27
+
28
+ ```python
29
+ from periplus_sdk import Client
30
+
31
+ with Client("http://localhost:8080") as client:
32
+ result = client.execute(
33
+ "SELECT capture_id, url FROM public_v1.captures LIMIT ?", [10]
34
+ )
35
+ print(result.columns, result.rows)
36
+ ```
37
+
38
+ Use `PERIPLUS_PUBLIC_URL` to omit the URL argument. No database credentials or
39
+ service token are needed for the public gateway. `AsyncClient` provides async
40
+ methods. `prepare` validates/explains SQL; `execute` returns typed columns, rows,
41
+ truncation and query metadata; `helpers` describes the installed public views.
42
+
43
+ The public schema is `public_v1`. HTML joins use `document_id` plus node index;
44
+ `document_id` identifies retained bytes and their HTML interpretation. Raw-byte hashes stay internal.
45
+ There is one query endpoint, with no experimental fallback. Client errors preserve
46
+ server categories and do not automatically retry executed queries.
47
+
48
+ For notebook/SQLAlchemy integration:
49
+
50
+ ```python
51
+ from periplus_sdk import sql_api
52
+ from sqlalchemy import text
53
+
54
+ engine = sql_api.create_engine(base_url="http://localhost:8080")
55
+ with engine.connect() as connection:
56
+ print(connection.execute(text("SELECT url FROM public_v1.captures LIMIT 5")).all())
57
+ engine.dispose()
58
+ ```
59
+
60
+ Marimo discovers the five public views through the helper catalogue. Column
61
+ reflection uses `SELECT * ... LIMIT 0` to retrieve native types without reading
62
+ corpus rows; no `SHOW`, `DESCRIBE`, or system-table access is required.
63
+
64
+ The DB-API connection advertises the ClickHouse dialect and converts native
65
+ nullable integer, decimal, date and datetime types. Nested types retain JSON wire
66
+ values. It is read-only: there are no client transactions or writable sessions.
67
+ Streaming cursors expose incomplete/truncated results explicitly; configure
68
+ `allow_partial` only when partial results suit the application.
69
+
70
+ See [the public schema](../../docs/SCHEMA.md) and [query boundary](../../docs/QUERY.md).
@@ -0,0 +1,52 @@
1
+ # Periplus Python SDK
2
+
3
+ Read-only access to the public ClickHouse corpus through the Periplus HTTP API.
4
+ Install the SDK from the same checkout as your deployment:
5
+
6
+ ```sh
7
+ python -m pip install ./packages/periplus-python-sdk
8
+ ```
9
+
10
+ ```python
11
+ from periplus_sdk import Client
12
+
13
+ with Client("http://localhost:8080") as client:
14
+ result = client.execute(
15
+ "SELECT capture_id, url FROM public_v1.captures LIMIT ?", [10]
16
+ )
17
+ print(result.columns, result.rows)
18
+ ```
19
+
20
+ Use `PERIPLUS_PUBLIC_URL` to omit the URL argument. No database credentials or
21
+ service token are needed for the public gateway. `AsyncClient` provides async
22
+ methods. `prepare` validates/explains SQL; `execute` returns typed columns, rows,
23
+ truncation and query metadata; `helpers` describes the installed public views.
24
+
25
+ The public schema is `public_v1`. HTML joins use `document_id` plus node index;
26
+ `document_id` identifies retained bytes and their HTML interpretation. Raw-byte hashes stay internal.
27
+ There is one query endpoint, with no experimental fallback. Client errors preserve
28
+ server categories and do not automatically retry executed queries.
29
+
30
+ For notebook/SQLAlchemy integration:
31
+
32
+ ```python
33
+ from periplus_sdk import sql_api
34
+ from sqlalchemy import text
35
+
36
+ engine = sql_api.create_engine(base_url="http://localhost:8080")
37
+ with engine.connect() as connection:
38
+ print(connection.execute(text("SELECT url FROM public_v1.captures LIMIT 5")).all())
39
+ engine.dispose()
40
+ ```
41
+
42
+ Marimo discovers the five public views through the helper catalogue. Column
43
+ reflection uses `SELECT * ... LIMIT 0` to retrieve native types without reading
44
+ corpus rows; no `SHOW`, `DESCRIBE`, or system-table access is required.
45
+
46
+ The DB-API connection advertises the ClickHouse dialect and converts native
47
+ nullable integer, decimal, date and datetime types. Nested types retain JSON wire
48
+ values. It is read-only: there are no client transactions or writable sessions.
49
+ Streaming cursors expose incomplete/truncated results explicitly; configure
50
+ `allow_partial` only when partial results suit the application.
51
+
52
+ See [the public schema](../../docs/SCHEMA.md) and [query boundary](../../docs/QUERY.md).
@@ -2,7 +2,7 @@
2
2
  license = "Apache-2.0"
3
3
  license-files = ["LICENSE", "NOTICE"]
4
4
  name = "periplus-python-sdk"
5
- version = "0.6.1"
5
+ version = "0.8.0"
6
6
  description = "Read-only Python client for the public Periplus query API"
7
7
  readme = "README.md"
8
8
  requires-python = ">=3.11"
@@ -0,0 +1,70 @@
1
+ Metadata-Version: 2.4
2
+ Name: periplus-python-sdk
3
+ Version: 0.8.0
4
+ Summary: Read-only Python client for the public Periplus query API
5
+ License-Expression: Apache-2.0
6
+ Project-URL: Repository, https://github.com/elei-io/periplus
7
+ Project-URL: Issues, https://github.com/elei-io/periplus/issues
8
+ Requires-Python: >=3.11
9
+ Description-Content-Type: text/markdown
10
+ License-File: LICENSE
11
+ License-File: NOTICE
12
+ Requires-Dist: httpx>=0.28
13
+ Requires-Dist: pydantic<3,>=2.12
14
+ Requires-Dist: sqlalchemy<3,>=2.0
15
+ Provides-Extra: notebook
16
+ Requires-Dist: marimo[sql]>=0.24.1; extra == "notebook"
17
+ Dynamic: license-file
18
+
19
+ # Periplus Python SDK
20
+
21
+ Read-only access to the public ClickHouse corpus through the Periplus HTTP API.
22
+ Install the SDK from the same checkout as your deployment:
23
+
24
+ ```sh
25
+ python -m pip install ./packages/periplus-python-sdk
26
+ ```
27
+
28
+ ```python
29
+ from periplus_sdk import Client
30
+
31
+ with Client("http://localhost:8080") as client:
32
+ result = client.execute(
33
+ "SELECT capture_id, url FROM public_v1.captures LIMIT ?", [10]
34
+ )
35
+ print(result.columns, result.rows)
36
+ ```
37
+
38
+ Use `PERIPLUS_PUBLIC_URL` to omit the URL argument. No database credentials or
39
+ service token are needed for the public gateway. `AsyncClient` provides async
40
+ methods. `prepare` validates/explains SQL; `execute` returns typed columns, rows,
41
+ truncation and query metadata; `helpers` describes the installed public views.
42
+
43
+ The public schema is `public_v1`. HTML joins use `document_id` plus node index;
44
+ `document_id` identifies retained bytes and their HTML interpretation. Raw-byte hashes stay internal.
45
+ There is one query endpoint, with no experimental fallback. Client errors preserve
46
+ server categories and do not automatically retry executed queries.
47
+
48
+ For notebook/SQLAlchemy integration:
49
+
50
+ ```python
51
+ from periplus_sdk import sql_api
52
+ from sqlalchemy import text
53
+
54
+ engine = sql_api.create_engine(base_url="http://localhost:8080")
55
+ with engine.connect() as connection:
56
+ print(connection.execute(text("SELECT url FROM public_v1.captures LIMIT 5")).all())
57
+ engine.dispose()
58
+ ```
59
+
60
+ Marimo discovers the five public views through the helper catalogue. Column
61
+ reflection uses `SELECT * ... LIMIT 0` to retrieve native types without reading
62
+ corpus rows; no `SHOW`, `DESCRIBE`, or system-table access is required.
63
+
64
+ The DB-API connection advertises the ClickHouse dialect and converts native
65
+ nullable integer, decimal, date and datetime types. Nested types retain JSON wire
66
+ values. It is read-only: there are no client transactions or writable sessions.
67
+ Streaming cursors expose incomplete/truncated results explicitly; configure
68
+ `allow_partial` only when partial results suit the application.
69
+
70
+ See [the public schema](../../docs/SCHEMA.md) and [query boundary](../../docs/QUERY.md).
@@ -15,8 +15,10 @@ src/periplus_sdk/errors.py
15
15
  src/periplus_sdk/py.typed
16
16
  src/periplus_sdk/sql_api.py
17
17
  src/periplus_sdk/sqlalchemy.py
18
+ src/periplus_sdk/stream.py
18
19
  src/periplus_sdk/types.py
19
20
  tests/test_client.py
20
21
  tests/test_dbapi.py
21
22
  tests/test_notebook.py
22
- tests/test_sql_api.py
23
+ tests/test_sql_api.py
24
+ tests/test_stream.py
@@ -79,10 +79,11 @@ def _payload(sql: str, parameters: Sequence[JsonValue] | None, schema_version: s
79
79
  class Client:
80
80
  """Reusable synchronous public query client. Close it or use a with block."""
81
81
 
82
- def __init__(self, base_url: str | None = None, *, timeout: float = 140, mode: Literal["stable", "experimental"] = "stable"):
83
- if mode not in {"stable", "experimental"}:
84
- raise ConfigurationError("mode must be stable or experimental.")
85
- self._query_path = "api/query/experimental/" if mode == "experimental" else "api/query/"
82
+ def __init__(self, base_url: str | None = None, *, timeout: float = 620, mode: Literal["stable"] = "stable"):
83
+ if mode not in {"stable"}:
84
+ raise ConfigurationError("Only the stable public catalogue is available.")
85
+ self.schema_version = "public_v1"
86
+ self._query_path = "api/query/"
86
87
  self._http = httpx.Client(**_options(base_url, timeout))
87
88
 
88
89
  def __enter__(self) -> Client:
@@ -101,11 +102,30 @@ class Client:
101
102
  raise TransportError("Could not complete the public query request.") from None
102
103
  return _decode(response, model)
103
104
 
104
- def prepare(self, sql: str, parameters: Sequence[JsonValue] | None = None, *, schema_version: str = "public_v1") -> PreparedQuery:
105
- return self._request("POST", "prep", PreparedQuery, json=_payload(sql, parameters, schema_version))
105
+ def prepare(self, sql: str, parameters: Sequence[JsonValue] | None = None, *, schema_version: str | None = None) -> PreparedQuery:
106
+ return self._request("POST", "prep", PreparedQuery, json=_payload(sql, parameters, schema_version if schema_version is not None else self.schema_version))
106
107
 
107
- def execute(self, sql: str, parameters: Sequence[JsonValue] | None = None, *, schema_version: str = "public_v1") -> QueryResult:
108
- return self._request("POST", "exec", QueryResult, json=_payload(sql, parameters, schema_version))
108
+ def execute(self, sql: str, parameters: Sequence[JsonValue] | None = None, *, schema_version: str | None = None) -> QueryResult:
109
+ return self._request("POST", "exec", QueryResult, json=_payload(sql, parameters, schema_version if schema_version is not None else self.schema_version))
110
+
111
+ def stream(self, sql: str, parameters: Sequence[JsonValue] | None = None, *,
112
+ schema_version: str | None = None, allow_partial: bool = False):
113
+ """Stream batches from one snapshot; use as a context manager for early exit."""
114
+ from .stream import MEDIA_TYPE, QueryStream
115
+ try:
116
+ request = self._http.build_request("POST", self._query_path + "exec",
117
+ headers={"accept": MEDIA_TYPE}, json=_payload(sql, parameters, schema_version if schema_version is not None else self.schema_version))
118
+ response = self._http.send(request, stream=True)
119
+ try:
120
+ if not response.is_success:
121
+ response.read()
122
+ _decode(response, QueryResult)
123
+ return QueryStream(response, allow_partial=allow_partial)
124
+ except BaseException:
125
+ response.close()
126
+ raise
127
+ except httpx.RequestError:
128
+ raise TransportError("Could not open the public query stream.") from None
109
129
 
110
130
  def helpers(self) -> QueryHelpers:
111
131
  return self._request("GET", "helpers", QueryHelpers)
@@ -114,10 +134,11 @@ class Client:
114
134
  class AsyncClient:
115
135
  """Reusable asynchronous public query client. Use an async with block."""
116
136
 
117
- def __init__(self, base_url: str | None = None, *, timeout: float = 140, mode: Literal["stable", "experimental"] = "stable"):
118
- if mode not in {"stable", "experimental"}:
119
- raise ConfigurationError("mode must be stable or experimental.")
120
- self._query_path = "api/query/experimental/" if mode == "experimental" else "api/query/"
137
+ def __init__(self, base_url: str | None = None, *, timeout: float = 620, mode: Literal["stable"] = "stable"):
138
+ if mode not in {"stable"}:
139
+ raise ConfigurationError("Only the stable public catalogue is available.")
140
+ self.schema_version = "public_v1"
141
+ self._query_path = "api/query/"
121
142
  self._http = httpx.AsyncClient(**_options(base_url, timeout))
122
143
 
123
144
  async def __aenter__(self) -> AsyncClient:
@@ -136,11 +157,11 @@ class AsyncClient:
136
157
  raise TransportError("Could not complete the public query request.") from None
137
158
  return _decode(response, model)
138
159
 
139
- async def prepare(self, sql: str, parameters: Sequence[JsonValue] | None = None, *, schema_version: str = "public_v1") -> PreparedQuery:
140
- return await self._request("POST", "prep", PreparedQuery, json=_payload(sql, parameters, schema_version))
160
+ async def prepare(self, sql: str, parameters: Sequence[JsonValue] | None = None, *, schema_version: str | None = None) -> PreparedQuery:
161
+ return await self._request("POST", "prep", PreparedQuery, json=_payload(sql, parameters, schema_version if schema_version is not None else self.schema_version))
141
162
 
142
- async def execute(self, sql: str, parameters: Sequence[JsonValue] | None = None, *, schema_version: str = "public_v1") -> QueryResult:
143
- return await self._request("POST", "exec", QueryResult, json=_payload(sql, parameters, schema_version))
163
+ async def execute(self, sql: str, parameters: Sequence[JsonValue] | None = None, *, schema_version: str | None = None) -> QueryResult:
164
+ return await self._request("POST", "exec", QueryResult, json=_payload(sql, parameters, schema_version if schema_version is not None else self.schema_version))
144
165
 
145
166
  async def helpers(self) -> QueryHelpers:
146
167
  return await self._request("GET", "helpers", QueryHelpers)
@@ -1,7 +1,7 @@
1
1
  """Read-only DB-API 2.0 connection over the public query API.
2
2
 
3
- Each execute is an independent server snapshot. Fetching consumes a bounded local
4
- result, never a remote cursor. Connections and cursors must not be shared by threads.
3
+ Each execute is an independent server snapshot. Fetching consumes a bounded stream.
4
+ Connections and cursors must not be shared by threads.
5
5
  """
6
6
  from __future__ import annotations
7
7
 
@@ -11,11 +11,10 @@ from collections.abc import Sequence
11
11
  from datetime import date, datetime, time
12
12
  from decimal import Decimal
13
13
  from typing import Any, Literal
14
- import warnings
15
14
 
16
15
  from .client import Client
17
16
  from .errors import ApiError, ConfigurationError, PeriplusError, ResponseError, TransportError
18
- from .types import QueryResult
17
+ from .stream import StreamResult
19
18
 
20
19
  apilevel = "2.0"
21
20
  threadsafety = 1
@@ -26,10 +25,6 @@ class Warning(builtins.Warning):
26
25
  """DB-API warning."""
27
26
 
28
27
 
29
- class TruncationWarning(Warning):
30
- """The server returned only part of the query result."""
31
-
32
-
33
28
  class Error(PeriplusError):
34
29
  """Base DB-API error; API failures preserve their safe error attributes."""
35
30
 
@@ -91,54 +86,47 @@ def TimestampFromTicks(ticks: float) -> datetime:
91
86
  return datetime.fromtimestamp(ticks)
92
87
 
93
88
 
94
- _INTEGER_TYPES = {"TINYINT", "SMALLINT", "INTEGER", "BIGINT", "HUGEINT", "UTINYINT",
95
- "USMALLINT", "UINTEGER", "UBIGINT", "UHUGEINT", "BIGNUM"}
96
- _FLOAT_TYPES = {"FLOAT", "DOUBLE", "REAL"}
97
- _TIME_TYPES = {"TIME", "TIME WITH TIME ZONE", "TIMETZ"}
98
- _TIMESTAMP_TYPES = {"TIMESTAMP", "TIMESTAMP_S", "TIMESTAMP_MS", "TIMESTAMP_NS",
99
- "TIMESTAMP WITH TIME ZONE", "TIMESTAMPTZ"}
89
+ _INTEGER_TYPES = {f'{prefix}Int{bits}' for prefix in ('', 'U') for bits in (8,16,32,64,128,256)}
90
+ _FLOAT_TYPES = {'Float32', 'Float64'}
91
+
92
+
93
+ def _base_type(value: str) -> str:
94
+ while value.startswith(('Nullable(', 'LowCardinality(')):
95
+ value = value[value.index('(')+1:-1]
96
+ return value
100
97
 
101
98
 
102
99
  class _TypeCategory:
103
- def __init__(self, names: set[str], prefix: str = ""):
104
- self.names, self.prefix = names, prefix
100
+ def __init__(self, names: set[str], prefixes: tuple[str, ...] = ()):
101
+ self.names, self.prefixes = names, prefixes
105
102
 
106
103
  def __eq__(self, other: object) -> bool:
107
- return isinstance(other, str) and (other in self.names or bool(self.prefix and other.startswith(self.prefix)))
104
+ if not isinstance(other, str):return False
105
+ value = _base_type(other)
106
+ return value in self.names or value.startswith(self.prefixes)
108
107
 
109
108
 
110
- STRING = _TypeCategory({"VARCHAR", "UUID", "JSON", "ENUM"})
111
- BINARY = _TypeCategory({"BLOB"})
112
- NUMBER = _TypeCategory(_INTEGER_TYPES | _FLOAT_TYPES | {"BOOLEAN"}, "DECIMAL(")
113
- DATETIME = _TypeCategory({"DATE"} | _TIME_TYPES | _TIMESTAMP_TYPES)
109
+ STRING = _TypeCategory({'String','UUID','JSON'}, ('FixedString(', 'Enum'))
110
+ BINARY = _TypeCategory(set())
111
+ NUMBER = _TypeCategory(_INTEGER_TYPES | _FLOAT_TYPES | {'Bool'}, ('Decimal',))
112
+ DATETIME = _TypeCategory({'Date','Date32'}, ('DateTime','Time'))
114
113
  ROWID = _TypeCategory(set())
115
114
 
116
115
 
117
116
  def _value(value: Any, sql_type: str) -> Any:
118
- if value is None:
119
- return None
120
- if sql_type in _INTEGER_TYPES:
121
- return int(value)
122
- if sql_type in _FLOAT_TYPES:
123
- return float(value)
124
- if sql_type.startswith("DECIMAL("):
125
- return Decimal(str(value))
126
- # Preserve infinities and out-of-range dates rather than clipping them.
127
- if sql_type == "DATE":
128
- try:
129
- return date.fromisoformat(value)
130
- except ValueError:
131
- return value
132
- if sql_type in _TIME_TYPES:
133
- return time.fromisoformat(value)
134
- if sql_type in _TIMESTAMP_TYPES:
135
- try:
136
- return datetime.fromisoformat(value)
137
- except ValueError:
138
- return value
139
- if sql_type == "BLOB":
140
- return base64.b64decode(value, validate=True)
141
- # UUIDs remain strings, and nested/other types retain their JSON wire values.
117
+ if value is None:return None
118
+ sql_type = _base_type(sql_type)
119
+ if sql_type in _INTEGER_TYPES:return int(value)
120
+ if sql_type in _FLOAT_TYPES:return float(value)
121
+ if sql_type.startswith('Decimal'):return Decimal(str(value))
122
+ if sql_type == 'Bool':return value in (True, 1, '1', 'true')
123
+ if sql_type in ('Date','Date32'):
124
+ try:return date.fromisoformat(value)
125
+ except ValueError:return value
126
+ if sql_type.startswith('DateTime'):
127
+ try:return datetime.fromisoformat(value)
128
+ except ValueError:return value
129
+ if sql_type.startswith('Time'):return time.fromisoformat(value)
142
130
  return value
143
131
 
144
132
 
@@ -147,24 +135,30 @@ def _parameter(value: Any) -> Any:
147
135
  return value
148
136
  if isinstance(value, (Decimal, date, time)):
149
137
  return str(value) if isinstance(value, Decimal) else value.isoformat()
150
- raise ProgrammingError("Parameters must be scalar JSON values, Decimal, date, time or datetime; use explicit SQL casts for typed strings.")
138
+ if isinstance(value, (list, tuple)):
139
+ if any(isinstance(item, (list, tuple, dict)) for item in value):
140
+ raise ProgrammingError("Collection parameters must be one-dimensional lists of scalar values.")
141
+ return [_parameter(item) for item in value]
142
+ raise ProgrammingError("Parameters must be scalar values or scalar lists; use explicit SQL casts for typed values.")
151
143
 
152
144
 
153
145
  class Connection:
154
146
  """Marimo-discoverable, read-only connection; commit is a no-op."""
155
147
 
156
- dialect = "duckdb"
148
+ dialect = "clickhouse"
157
149
 
158
- def __init__(self, base_url: str | None = None, *, timeout: float = 140,
159
- mode: Literal["stable", "experimental"] = "stable",
160
- schema_version: str = "public_v1"):
150
+ def __init__(self, base_url: str | None = None, *, timeout: float = 620,
151
+ mode: Literal["stable"] = "stable",
152
+ schema_version: str | None = None, allow_partial: bool = False):
161
153
  try:
162
154
  self._client = Client(base_url, timeout=timeout, mode=mode)
163
155
  except ConfigurationError as exc:
164
156
  raise InterfaceError(str(exc)) from exc
165
- self.schema_version = schema_version
157
+ self.schema_version = schema_version if schema_version is not None else ("public_v1")
158
+ self.allow_partial = allow_partial
159
+ self._cursors = set()
166
160
  self.closed = False
167
- self.last_result: QueryResult | None = None
161
+ self.last_result: StreamResult | None = None
168
162
 
169
163
  def _check(self) -> None:
170
164
  if self.closed:
@@ -172,7 +166,9 @@ class Connection:
172
166
 
173
167
  def cursor(self) -> Cursor:
174
168
  self._check()
175
- return Cursor(self)
169
+ cursor = Cursor(self)
170
+ self._cursors.add(cursor)
171
+ return cursor
176
172
 
177
173
  def execute(self, operation: str, parameters: Sequence[Any] | None = None) -> Cursor:
178
174
  cursor = self.cursor()
@@ -192,6 +188,8 @@ class Connection:
192
188
 
193
189
  def close(self) -> None:
194
190
  if not self.closed:
191
+ for cursor in list(self._cursors):
192
+ cursor.close()
195
193
  self._client.close()
196
194
  self.closed = True
197
195
  self.last_result = None
@@ -204,25 +202,26 @@ class Connection:
204
202
  self.close()
205
203
 
206
204
 
207
- def connect(base_url: str | None = None, *, timeout: float = 140,
208
- mode: Literal["stable", "experimental"] = "stable",
209
- schema_version: str = "public_v1") -> Connection:
210
- return Connection(base_url, timeout=timeout, mode=mode, schema_version=schema_version)
205
+ def connect(base_url: str | None = None, *, timeout: float = 620,
206
+ mode: Literal["stable"] = "stable",
207
+ schema_version: str | None = None, allow_partial: bool = False) -> Connection:
208
+ return Connection(base_url, timeout=timeout, mode=mode, schema_version=schema_version, allow_partial=allow_partial)
211
209
 
212
210
 
213
211
  class Cursor:
214
- """A buffered result. Metadata stays on result after rows are consumed."""
212
+ """Incremental cursor. Metadata stays available without retaining consumed rows."""
215
213
 
216
214
  arraysize = 1
217
215
 
218
216
  def __init__(self, connection: Connection):
219
217
  self.connection = connection
220
218
  self.closed = False
221
- self.result: QueryResult | None = None
219
+ self.result: StreamResult | None = None
222
220
  self.description: list[tuple[Any, ...]] | None = None
223
221
  self.rowcount = -1
224
222
  self._rows: list[tuple[Any, ...]] = []
225
223
  self._position = 0
224
+ self._stream = None
226
225
 
227
226
  def _check(self, *, result: bool = False) -> None:
228
227
  self.connection._check()
@@ -233,6 +232,9 @@ class Cursor:
233
232
 
234
233
  def execute(self, operation: str, parameters: Sequence[Any] | None = None) -> Cursor:
235
234
  self._check()
235
+ if self._stream is not None:
236
+ self._stream.close()
237
+ self._stream = None
236
238
  self.result, self.description, self.rowcount = None, None, -1
237
239
  self._rows, self._position = [], 0
238
240
  self.connection.last_result = None
@@ -242,7 +244,9 @@ class Cursor:
242
244
  raise ProgrammingError("Use a positional parameter sequence with ? placeholders.")
243
245
  values = [_parameter(v) for v in parameters] if parameters is not None else []
244
246
  try:
245
- result = self.connection._client.execute(operation, values, schema_version=self.connection.schema_version)
247
+ self._stream = self.connection._client.stream(operation, values, schema_version=self.connection.schema_version,
248
+ allow_partial=self.connection.allow_partial)
249
+ result = self._stream.result
246
250
  except ApiError as exc:
247
251
  error = ProgrammingError if exc.code == "sql_invalid" else OperationalError
248
252
  raise error(str(exc), status_code=exc.status_code, code=exc.code,
@@ -251,26 +255,37 @@ class Cursor:
251
255
  raise OperationalError(str(exc)) from exc
252
256
  except ResponseError as exc:
253
257
  raise InterfaceError(str(exc)) from exc
254
- if len(result.columns) != len(result.types) or any(len(r) != len(result.columns) for r in result.rows):
255
- raise InterfaceError("Query columns, types and rows have inconsistent widths.")
256
- try:
257
- rows = [tuple(_value(v, t) for v, t in zip(row, result.types, strict=True)) for row in result.rows]
258
- except (ValueError, TypeError, ArithmeticError) as exc:
259
- raise DataError("Query value does not match its SQL type.") from exc
260
258
  self.result = self.connection.last_result = result
261
259
  self.description = [(name, kind, None, None, None, None, None)
262
260
  for name, kind in zip(result.columns, result.types, strict=True)]
263
- self._rows = rows
264
- self.rowcount = -1 if result.truncated else len(rows)
265
- if result.truncated:
266
- warnings.warn(f"Periplus returned a truncated result ({len(rows)} rows); inspect connection.last_result or cursor.result. Fetching does not retrieve additional rows.",
267
- TruncationWarning, stacklevel=2)
268
261
  return self
269
262
 
263
+ def _batch(self):
264
+ try:
265
+ batch = next(self._stream)
266
+ self._rows = [tuple(_value(v, t) for v, t in zip(row, self.result.types, strict=True)) for row in batch]
267
+ self._position = 0
268
+ return True
269
+ except StopIteration:
270
+ self.rowcount = -1 if self.result.truncated else self.result.row_count
271
+ self._rows, self._position = [], 0
272
+ return False
273
+ except ApiError as exc:
274
+ raise OperationalError(str(exc), status_code=exc.status_code, code=exc.code,
275
+ retry_after_seconds=exc.retry_after_seconds) from exc
276
+ except TransportError as exc:
277
+ raise OperationalError(str(exc)) from exc
278
+ except ResponseError as exc:
279
+ raise InterfaceError(str(exc)) from exc
280
+ except (ValueError, TypeError, ArithmeticError) as exc:
281
+ self._stream.close()
282
+ raise DataError("Query value does not match its SQL type.") from exc
283
+
270
284
  def fetchone(self) -> tuple[Any, ...] | None:
271
285
  self._check(result=True)
272
- if self._position == len(self._rows):
273
- return None
286
+ while self._position == len(self._rows):
287
+ if not self._batch():
288
+ return None
274
289
  row = self._rows[self._position]
275
290
  self._position += 1
276
291
  return row
@@ -280,14 +295,17 @@ class Cursor:
280
295
  size = self.arraysize if size is None else size
281
296
  if not isinstance(size, int) or size < 0:
282
297
  raise ProgrammingError("Fetch size must be a non-negative integer.")
283
- end = min(self._position + size, len(self._rows))
284
- rows = self._rows[self._position:end]
285
- self._position = end
298
+ rows = []
299
+ for _ in range(size):
300
+ row = self.fetchone()
301
+ if row is None:
302
+ break
303
+ rows.append(row)
286
304
  return rows
287
305
 
288
306
  def fetchall(self) -> list[tuple[Any, ...]]:
289
307
  self._check(result=True)
290
- return self.fetchmany(len(self._rows) - self._position)
308
+ return list(self)
291
309
 
292
310
  def executemany(self, operation: str, seq_of_parameters: Any) -> None:
293
311
  self._check()
@@ -300,6 +318,9 @@ class Cursor:
300
318
  self._check()
301
319
 
302
320
  def close(self) -> None:
321
+ if self._stream is not None:
322
+ self._stream.close()
323
+ self.connection._cursors.discard(self)
303
324
  self.closed = True
304
325
  self._rows = []
305
326
  self.result = None
@@ -0,0 +1,47 @@
1
+ """Notebook-friendly SQL engine over the public Periplus query API."""
2
+ from __future__ import annotations
3
+
4
+ from typing import Literal
5
+
6
+ from sqlalchemy import create_engine as _create_engine, event
7
+ from sqlalchemy.engine import Engine, URL
8
+
9
+
10
+ def create_engine(
11
+ base_url: str | None = None,
12
+ *,
13
+ mode: Literal["stable"] = "stable",
14
+ timeout: float = 620,
15
+ schema_version: str | None = None,
16
+ allow_partial: bool = False,
17
+ ) -> Engine:
18
+ """Create a SQLAlchemy engine recognized by marimo and other SQL tools.
19
+
20
+ The public URL defaults to PERIPLUS_PUBLIC_URL. Connections are opened lazily;
21
+ dispose the engine when finished. Each query uses an independent server snapshot.
22
+ """
23
+ engine = _create_engine(
24
+ URL.create("periplus", database="periplus"),
25
+ connect_args={"base_url": base_url, "mode": mode, "timeout": timeout, "schema_version": schema_version, "allow_partial": allow_partial},
26
+ )
27
+
28
+ event.listen(engine, "before_execute", _parameters, retval=True)
29
+ return engine
30
+
31
+
32
+ def _parameters(connection, clauseelement, multiparams, params, execution_options):
33
+ bindings = execution_options.get("periplus_parameters", {})
34
+ if bindings and not multiparams:
35
+ names = clauseelement.compile().params
36
+ params = {**{name: value for name, value in bindings.items() if name in names}, **params}
37
+ return clauseelement, multiparams, params
38
+
39
+
40
+ def bind(engine: Engine, **parameters) -> Engine:
41
+ """Create a marimo-discoverable engine with named SQL parameters for this cell.
42
+
43
+ Uses the original engine's pool. No query, upload, or server state is created.
44
+ Write :name placeholders in SQL cells; lists can be CAST(:ids AS VARCHAR[]).
45
+ """
46
+ from .dbapi import _parameter
47
+ return engine.execution_options(periplus_parameters={name: _parameter(value) for name, value in parameters.items()})