periplus-python-sdk 0.9.0__tar.gz → 0.12.1__tar.gz

This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
Files changed (39) hide show
  1. periplus_python_sdk-0.12.1/PKG-INFO +186 -0
  2. periplus_python_sdk-0.12.1/README.md +166 -0
  3. {periplus_python_sdk-0.9.0 → periplus_python_sdk-0.12.1}/licensing/THIRD_PARTY_NOTICES.md +1 -1
  4. {periplus_python_sdk-0.9.0 → periplus_python_sdk-0.12.1}/pyproject.toml +2 -2
  5. periplus_python_sdk-0.12.1/src/periplus_python_sdk.egg-info/PKG-INFO +186 -0
  6. {periplus_python_sdk-0.9.0 → periplus_python_sdk-0.12.1}/src/periplus_python_sdk.egg-info/SOURCES.txt +5 -0
  7. periplus_python_sdk-0.12.1/src/periplus_sdk/__init__.py +13 -0
  8. periplus_python_sdk-0.12.1/src/periplus_sdk/async_resources.py +366 -0
  9. periplus_python_sdk-0.12.1/src/periplus_sdk/async_stream.py +104 -0
  10. periplus_python_sdk-0.12.1/src/periplus_sdk/client.py +325 -0
  11. {periplus_python_sdk-0.9.0 → periplus_python_sdk-0.12.1}/src/periplus_sdk/errors.py +3 -1
  12. periplus_python_sdk-0.12.1/src/periplus_sdk/resources.py +367 -0
  13. periplus_python_sdk-0.12.1/src/periplus_sdk/resources_types.py +348 -0
  14. {periplus_python_sdk-0.9.0 → periplus_python_sdk-0.12.1}/src/periplus_sdk/sqlalchemy.py +25 -25
  15. {periplus_python_sdk-0.9.0 → periplus_python_sdk-0.12.1}/src/periplus_sdk/stream.py +8 -2
  16. periplus_python_sdk-0.12.1/src/periplus_sdk/types.py +147 -0
  17. {periplus_python_sdk-0.9.0 → periplus_python_sdk-0.12.1}/tests/test_client.py +66 -4
  18. {periplus_python_sdk-0.9.0 → periplus_python_sdk-0.12.1}/tests/test_dbapi.py +2 -2
  19. {periplus_python_sdk-0.9.0 → periplus_python_sdk-0.12.1}/tests/test_notebook.py +11 -7
  20. periplus_python_sdk-0.12.1/tests/test_platform.py +210 -0
  21. {periplus_python_sdk-0.9.0 → periplus_python_sdk-0.12.1}/tests/test_stream.py +7 -1
  22. periplus_python_sdk-0.9.0/PKG-INFO +0 -83
  23. periplus_python_sdk-0.9.0/README.md +0 -63
  24. periplus_python_sdk-0.9.0/src/periplus_python_sdk.egg-info/PKG-INFO +0 -83
  25. periplus_python_sdk-0.9.0/src/periplus_sdk/__init__.py +0 -9
  26. periplus_python_sdk-0.9.0/src/periplus_sdk/client.py +0 -172
  27. periplus_python_sdk-0.9.0/src/periplus_sdk/types.py +0 -59
  28. {periplus_python_sdk-0.9.0 → periplus_python_sdk-0.12.1}/licensing/LICENSE +0 -0
  29. {periplus_python_sdk-0.9.0 → periplus_python_sdk-0.12.1}/licensing/NOTICE +0 -0
  30. {periplus_python_sdk-0.9.0 → periplus_python_sdk-0.12.1}/licensing/README.md +0 -0
  31. {periplus_python_sdk-0.9.0 → periplus_python_sdk-0.12.1}/setup.cfg +0 -0
  32. {periplus_python_sdk-0.9.0 → periplus_python_sdk-0.12.1}/src/periplus_python_sdk.egg-info/dependency_links.txt +0 -0
  33. {periplus_python_sdk-0.9.0 → periplus_python_sdk-0.12.1}/src/periplus_python_sdk.egg-info/entry_points.txt +0 -0
  34. {periplus_python_sdk-0.9.0 → periplus_python_sdk-0.12.1}/src/periplus_python_sdk.egg-info/requires.txt +0 -0
  35. {periplus_python_sdk-0.9.0 → periplus_python_sdk-0.12.1}/src/periplus_python_sdk.egg-info/top_level.txt +0 -0
  36. {periplus_python_sdk-0.9.0 → periplus_python_sdk-0.12.1}/src/periplus_sdk/dbapi.py +0 -0
  37. {periplus_python_sdk-0.9.0 → periplus_python_sdk-0.12.1}/src/periplus_sdk/py.typed +0 -0
  38. {periplus_python_sdk-0.9.0 → periplus_python_sdk-0.12.1}/src/periplus_sdk/sql_api.py +0 -0
  39. {periplus_python_sdk-0.9.0 → periplus_python_sdk-0.12.1}/tests/test_sql_api.py +0 -0
@@ -0,0 +1,186 @@
1
+ Metadata-Version: 2.4
2
+ Name: periplus-python-sdk
3
+ Version: 0.12.1
4
+ Summary: Typed Periplus platform client with SQL and notebook integration
5
+ License-Expression: AGPL-3.0-only
6
+ Project-URL: Repository, https://github.com/elei-io/periplus
7
+ Project-URL: Issues, https://github.com/elei-io/periplus/issues
8
+ Requires-Python: >=3.11
9
+ Description-Content-Type: text/markdown
10
+ License-File: licensing/LICENSE
11
+ License-File: licensing/NOTICE
12
+ License-File: licensing/README.md
13
+ License-File: licensing/THIRD_PARTY_NOTICES.md
14
+ Requires-Dist: httpx>=0.28
15
+ Requires-Dist: pydantic<3,>=2.12
16
+ Requires-Dist: sqlalchemy<3,>=2.0
17
+ Provides-Extra: notebook
18
+ Requires-Dist: marimo[sql]>=0.24.1; extra == "notebook"
19
+ Dynamic: license-file
20
+
21
+ # Periplus Python SDK
22
+
23
+ Typed organization operations and SQL access through the Periplus HTTP API.
24
+ The 0.12.0 platform interface requires the matching API release. It replaces the
25
+ discovery, retention and monitoring namespaces with captures, crawls, pins and
26
+ schedules, and includes the source-snapshot metadata introduced in 0.10.0.
27
+ Install from PyPI:
28
+
29
+ ```sh
30
+ python -m pip install --upgrade periplus-python-sdk
31
+ ```
32
+
33
+ ```python
34
+ from periplus_sdk import Client
35
+
36
+ with Client("https://api.periplus.dev", api_key="ppl_…") as client:
37
+ result = client.execute(
38
+ "SELECT capture_id, url FROM public_v1.captures LIMIT ?", [10]
39
+ )
40
+ print(result.columns, result.rows)
41
+ ```
42
+
43
+ The default origin is `https://api.periplus.dev`. Use `PERIPLUS_API_URL` to override it and `PERIPLUS_API_KEY` to
44
+ supply your personal organization API key. A key with `organization:sql:exec`
45
+ is required. Database usernames/passwords and anonymous access are not supported.
46
+ Use HTTPS outside loopback development. Connect directly to the API origin,
47
+ not the public marketing site.
48
+
49
+ Version 0.9.0 requires personal API keys instead of database credentials. Create
50
+ a Developer key in the app's Access → SDK & API. It inherits your current access in
51
+ that organization; SQL requires your membership to have `organization:sql:exec`.
52
+ For local development, install `./clients/periplus-python-sdk` from the repository
53
+ root and connect to `http://localhost:8000`.
54
+
55
+ `AsyncClient` accepts the same options. `prepare` explains a SELECT; `execute`
56
+ returns typed columns/rows for read-only queries. `schema()` returns
57
+ visible tables, column types/descriptions and helper documentation. All SQL uses
58
+ `POST /api/v1/sql`; schema discovery uses `GET /api/v1/schema`. ClickHouse enforces
59
+ permissions. Buffered SQL reads retry HTTP 502, 503 and 504 up to twice with bounded
60
+ backoff starting in 0.12.1. Set `sql_retries=0` to disable this, or choose up to three retries. Each
61
+ attempt may observe a newer corpus snapshot. Mutations, transport failures and
62
+ streamed queries are not retried automatically.
63
+
64
+ Public HTML joins use `capture_id` and `node_index`; `document_id` identifies exact
65
+ raw bytes. Public shorthand uses the `public_v1` schema.
66
+
67
+ In 0.10.0, `result.source_snapshot` is a typed `SourceSnapshot` containing
68
+ `layout_id` (UUID) and `publication_epoch` (integer), or `None` when no build-bound
69
+ public corpus was read. Buffered, asynchronous, streamed and DB-API results share
70
+ this contract. Compare both fields, not just the epoch. The identity describes
71
+ the public corpus inputs, not any native staff/external inputs. It does not retain
72
+ the data or request historical reads. This replaces the old nullable integer field;
73
+ use this SDK version with the corresponding API release.
74
+
75
+ For notebook/SQLAlchemy integration:
76
+
77
+ ```python
78
+ from periplus_sdk import sql_api
79
+ from sqlalchemy import text
80
+
81
+ engine = sql_api.create_engine(base_url="http://localhost:8000", api_key="ppl_…")
82
+ with engine.connect() as connection:
83
+ print(connection.execute(text("SELECT url FROM public_v1.captures LIMIT 5")).all())
84
+ engine.dispose()
85
+ ```
86
+
87
+ Marimo discovers accessible tables, views, typed columns and comments through the
88
+ schema endpoint. Reflection does not execute SQL. One SQLAlchemy Inspector caches
89
+ its metadata; call `inspector.clear_cache()` to refresh it. Missing metadata raises
90
+ an error rather than presenting an apparently complete empty schema.
91
+
92
+ The DB-API connection advertises the ClickHouse dialect and converts native
93
+ nullable integer, decimal, date and datetime types. Nested types retain JSON wire
94
+ values. Writes are not exposed through the query API. There are no client
95
+ transactions; each statement is independent.
96
+ Streaming cursors expose incomplete/truncated results explicitly; configure
97
+ `allow_partial` only when partial results suit the application.
98
+
99
+ See [the public schema](../../docs/SCHEMA.md) and [query boundary](../../docs/QUERY.md).
100
+
101
+ ## Organization resources
102
+
103
+ A Developer key selects your organization and uses your live membership access. Agent keys authenticate only to the hosted MCP endpoint, not this SDK.
104
+ `Client` and `AsyncClient` expose the same namespaces:
105
+
106
+ | Namespace | Operations |
107
+ | --- | --- |
108
+ | `identity`, `availability` | `get` |
109
+ | `captures`, `crawls` | `create`, `get`, `list`, `iter`, `pages`, `cancel` |
110
+ | `pins` | `create`, `get`, `list`, `iter`, `captures`, `update`, `delete` |
111
+ | `schedules` | `create`, `get`, `list`, `iter`, `pages`, `captures`, `update`, `pause`, `resume`, `delete` |
112
+ | `saved_queries` | `create`, `get`, `list`, `iter`, `rename`, `delete` |
113
+ | `members` | `list`, `update_role`, `remove` |
114
+ | `invitations` | `list`, `create`, `cancel` |
115
+ | `api_keys` | `create`, `list`, `iter`, `revoke` |
116
+ | `usage` | `get` |
117
+ | `query_history` | `list`, `iter`, `get`, `summary` |
118
+ | `audit` | `list`, `iter` |
119
+
120
+ Choose the resource by what you know and how long you need it:
121
+
122
+ | Resource | Use it when |
123
+ | --- | --- |
124
+ | `captures` | You know the URLs (1–100) and need them in SQL now |
125
+ | `crawls` | You know a starting point, not the URLs |
126
+ | `pins` | You need exact captures for longer than seven days |
127
+ | `schedules` | You need the same pages again later |
128
+
129
+ ```python
130
+ from periplus_sdk import Client
131
+
132
+ with Client() as client:
133
+ capture = client.captures.create(urls=["https://example.com/pricing"])
134
+ capture = client.captures.get(capture, wait_seconds=30)
135
+ if capture.finished and capture.queryable:
136
+ for page in client.captures.pages(capture).items:
137
+ print(page.url, page.status, page.capture_id)
138
+
139
+ crawl = client.crawls.create(seeds=["https://example.com/docs/"], page_limit=200,
140
+ allowed_paths=["/docs/*"])
141
+
142
+ pin = client.pins.create(
143
+ name="Research sources", days=90,
144
+ sql="SELECT capture_id FROM captures WHERE domain(url) = ?",
145
+ parameters=["example.com"],
146
+ )
147
+ schedule = client.schedules.create(name="Pricing", every_days=7,
148
+ urls=["https://example.com/pricing"])
149
+ print(pin.captures, schedule.projected_pages_per_month)
150
+ ```
151
+
152
+ Every capture is kept for seven days after collection unless a pin keeps it. A
153
+ capture always fetches again and never returns an existing capture, so check
154
+ coverage with SQL first. `request_id` (a capture or crawl `id`) is what you
155
+ ordered; each page's `capture_id` is what the crawler produced and is the SQL key.
156
+ `get(..., wait_seconds=N)` returns as soon as the request is finished and queryable,
157
+ or after at most 30 seconds; the SDK never polls on its own.
158
+
159
+ Pins and schedules accept explicit IDs/URLs or one read-only SQL query with
160
+ parameters, and are active immediately. SQL runs once at creation and its exact
161
+ result is locked in; running the same query beforehand is only an estimate.
162
+ Schedule URLs need not be in the corpus. `pin_days` on a capture, crawl or schedule
163
+ pins each successful capture it produces and requires `organization:pins:write`.
164
+ `pins.update` changes `days` (moving every capture's expiry by the difference) or
165
+ the name; `schedules.update` changes `every_days` or the name. Both accept a fetched
166
+ object or an ID with `expected_version`; conflicts are never retried. Deletion needs
167
+ no version: deleting a pin removes future protection, not captures.
168
+
169
+ Paginated lists return `Page[T]` (`items`, `next_cursor`). Pass cursors unchanged
170
+ or use `iter()`; membership/invitation lists are bounded snapshots instead.
171
+ Offset-based lists can shift during concurrent changes. Query history, request
172
+ page feeds and scheduled capture feeds use opaque keyset cursors. UTC usage ranges have an exclusive end date, at most 93 days. Usage reports pages
173
+ captured (broken down by capture, crawl or schedule), challenge-resolution pages and
174
+ pinned capture-days.
175
+ Query history is best-effort and expires after 30 days; it is not a billing ledger.
176
+
177
+ `ApiError` includes HTTP status, code, optional request ID, validation fields and
178
+ Retry-After seconds. `TransportError` means the outcome of a write can be unknown.
179
+ Keep creation IDs to reconcile; never blindly retry. Key creation is one-time:
180
+ `created.secret.get_secret_value()` reveals the secret and must only be used for
181
+ secure storage. Its ordinary representation is masked. Lost secrets cannot be
182
+ recovered.
183
+
184
+ For async streaming, use `async with await client.stream(sql) as stream` followed
185
+ by `async for batch in stream`. Streams validate completion and close on early
186
+ exit, errors, and cancellation. No threads or background polling are introduced.
@@ -0,0 +1,166 @@
1
+ # Periplus Python SDK
2
+
3
+ Typed organization operations and SQL access through the Periplus HTTP API.
4
+ The 0.12.0 platform interface requires the matching API release. It replaces the
5
+ discovery, retention and monitoring namespaces with captures, crawls, pins and
6
+ schedules, and includes the source-snapshot metadata introduced in 0.10.0.
7
+ Install from PyPI:
8
+
9
+ ```sh
10
+ python -m pip install --upgrade periplus-python-sdk
11
+ ```
12
+
13
+ ```python
14
+ from periplus_sdk import Client
15
+
16
+ with Client("https://api.periplus.dev", api_key="ppl_…") as client:
17
+ result = client.execute(
18
+ "SELECT capture_id, url FROM public_v1.captures LIMIT ?", [10]
19
+ )
20
+ print(result.columns, result.rows)
21
+ ```
22
+
23
+ The default origin is `https://api.periplus.dev`. Use `PERIPLUS_API_URL` to override it and `PERIPLUS_API_KEY` to
24
+ supply your personal organization API key. A key with `organization:sql:exec`
25
+ is required. Database usernames/passwords and anonymous access are not supported.
26
+ Use HTTPS outside loopback development. Connect directly to the API origin,
27
+ not the public marketing site.
28
+
29
+ Version 0.9.0 requires personal API keys instead of database credentials. Create
30
+ a Developer key in the app's Access → SDK & API. It inherits your current access in
31
+ that organization; SQL requires your membership to have `organization:sql:exec`.
32
+ For local development, install `./clients/periplus-python-sdk` from the repository
33
+ root and connect to `http://localhost:8000`.
34
+
35
+ `AsyncClient` accepts the same options. `prepare` explains a SELECT; `execute`
36
+ returns typed columns/rows for read-only queries. `schema()` returns
37
+ visible tables, column types/descriptions and helper documentation. All SQL uses
38
+ `POST /api/v1/sql`; schema discovery uses `GET /api/v1/schema`. ClickHouse enforces
39
+ permissions. Buffered SQL reads retry HTTP 502, 503 and 504 up to twice with bounded
40
+ backoff starting in 0.12.1. Set `sql_retries=0` to disable this, or choose up to three retries. Each
41
+ attempt may observe a newer corpus snapshot. Mutations, transport failures and
42
+ streamed queries are not retried automatically.
43
+
44
+ Public HTML joins use `capture_id` and `node_index`; `document_id` identifies exact
45
+ raw bytes. Public shorthand uses the `public_v1` schema.
46
+
47
+ In 0.10.0, `result.source_snapshot` is a typed `SourceSnapshot` containing
48
+ `layout_id` (UUID) and `publication_epoch` (integer), or `None` when no build-bound
49
+ public corpus was read. Buffered, asynchronous, streamed and DB-API results share
50
+ this contract. Compare both fields, not just the epoch. The identity describes
51
+ the public corpus inputs, not any native staff/external inputs. It does not retain
52
+ the data or request historical reads. This replaces the old nullable integer field;
53
+ use this SDK version with the corresponding API release.
54
+
55
+ For notebook/SQLAlchemy integration:
56
+
57
+ ```python
58
+ from periplus_sdk import sql_api
59
+ from sqlalchemy import text
60
+
61
+ engine = sql_api.create_engine(base_url="http://localhost:8000", api_key="ppl_…")
62
+ with engine.connect() as connection:
63
+ print(connection.execute(text("SELECT url FROM public_v1.captures LIMIT 5")).all())
64
+ engine.dispose()
65
+ ```
66
+
67
+ Marimo discovers accessible tables, views, typed columns and comments through the
68
+ schema endpoint. Reflection does not execute SQL. One SQLAlchemy Inspector caches
69
+ its metadata; call `inspector.clear_cache()` to refresh it. Missing metadata raises
70
+ an error rather than presenting an apparently complete empty schema.
71
+
72
+ The DB-API connection advertises the ClickHouse dialect and converts native
73
+ nullable integer, decimal, date and datetime types. Nested types retain JSON wire
74
+ values. Writes are not exposed through the query API. There are no client
75
+ transactions; each statement is independent.
76
+ Streaming cursors expose incomplete/truncated results explicitly; configure
77
+ `allow_partial` only when partial results suit the application.
78
+
79
+ See [the public schema](../../docs/SCHEMA.md) and [query boundary](../../docs/QUERY.md).
80
+
81
+ ## Organization resources
82
+
83
+ A Developer key selects your organization and uses your live membership access. Agent keys authenticate only to the hosted MCP endpoint, not this SDK.
84
+ `Client` and `AsyncClient` expose the same namespaces:
85
+
86
+ | Namespace | Operations |
87
+ | --- | --- |
88
+ | `identity`, `availability` | `get` |
89
+ | `captures`, `crawls` | `create`, `get`, `list`, `iter`, `pages`, `cancel` |
90
+ | `pins` | `create`, `get`, `list`, `iter`, `captures`, `update`, `delete` |
91
+ | `schedules` | `create`, `get`, `list`, `iter`, `pages`, `captures`, `update`, `pause`, `resume`, `delete` |
92
+ | `saved_queries` | `create`, `get`, `list`, `iter`, `rename`, `delete` |
93
+ | `members` | `list`, `update_role`, `remove` |
94
+ | `invitations` | `list`, `create`, `cancel` |
95
+ | `api_keys` | `create`, `list`, `iter`, `revoke` |
96
+ | `usage` | `get` |
97
+ | `query_history` | `list`, `iter`, `get`, `summary` |
98
+ | `audit` | `list`, `iter` |
99
+
100
+ Choose the resource by what you know and how long you need it:
101
+
102
+ | Resource | Use it when |
103
+ | --- | --- |
104
+ | `captures` | You know the URLs (1–100) and need them in SQL now |
105
+ | `crawls` | You know a starting point, not the URLs |
106
+ | `pins` | You need exact captures for longer than seven days |
107
+ | `schedules` | You need the same pages again later |
108
+
109
+ ```python
110
+ from periplus_sdk import Client
111
+
112
+ with Client() as client:
113
+ capture = client.captures.create(urls=["https://example.com/pricing"])
114
+ capture = client.captures.get(capture, wait_seconds=30)
115
+ if capture.finished and capture.queryable:
116
+ for page in client.captures.pages(capture).items:
117
+ print(page.url, page.status, page.capture_id)
118
+
119
+ crawl = client.crawls.create(seeds=["https://example.com/docs/"], page_limit=200,
120
+ allowed_paths=["/docs/*"])
121
+
122
+ pin = client.pins.create(
123
+ name="Research sources", days=90,
124
+ sql="SELECT capture_id FROM captures WHERE domain(url) = ?",
125
+ parameters=["example.com"],
126
+ )
127
+ schedule = client.schedules.create(name="Pricing", every_days=7,
128
+ urls=["https://example.com/pricing"])
129
+ print(pin.captures, schedule.projected_pages_per_month)
130
+ ```
131
+
132
+ Every capture is kept for seven days after collection unless a pin keeps it. A
133
+ capture always fetches again and never returns an existing capture, so check
134
+ coverage with SQL first. `request_id` (a capture or crawl `id`) is what you
135
+ ordered; each page's `capture_id` is what the crawler produced and is the SQL key.
136
+ `get(..., wait_seconds=N)` returns as soon as the request is finished and queryable,
137
+ or after at most 30 seconds; the SDK never polls on its own.
138
+
139
+ Pins and schedules accept explicit IDs/URLs or one read-only SQL query with
140
+ parameters, and are active immediately. SQL runs once at creation and its exact
141
+ result is locked in; running the same query beforehand is only an estimate.
142
+ Schedule URLs need not be in the corpus. `pin_days` on a capture, crawl or schedule
143
+ pins each successful capture it produces and requires `organization:pins:write`.
144
+ `pins.update` changes `days` (moving every capture's expiry by the difference) or
145
+ the name; `schedules.update` changes `every_days` or the name. Both accept a fetched
146
+ object or an ID with `expected_version`; conflicts are never retried. Deletion needs
147
+ no version: deleting a pin removes future protection, not captures.
148
+
149
+ Paginated lists return `Page[T]` (`items`, `next_cursor`). Pass cursors unchanged
150
+ or use `iter()`; membership/invitation lists are bounded snapshots instead.
151
+ Offset-based lists can shift during concurrent changes. Query history, request
152
+ page feeds and scheduled capture feeds use opaque keyset cursors. UTC usage ranges have an exclusive end date, at most 93 days. Usage reports pages
153
+ captured (broken down by capture, crawl or schedule), challenge-resolution pages and
154
+ pinned capture-days.
155
+ Query history is best-effort and expires after 30 days; it is not a billing ledger.
156
+
157
+ `ApiError` includes HTTP status, code, optional request ID, validation fields and
158
+ Retry-After seconds. `TransportError` means the outcome of a write can be unknown.
159
+ Keep creation IDs to reconcile; never blindly retry. Key creation is one-time:
160
+ `created.secret.get_secret_value()` reveals the secret and must only be used for
161
+ secure storage. Its ordinary representation is masked. Lost secrets cannot be
162
+ recovered.
163
+
164
+ For async streaming, use `async with await client.stream(sql) as stream` followed
165
+ by `async for batch in stream`. Streams validate completion and close on early
166
+ exit, errors, and cancellation. No threads or background polling are introduced.
@@ -7,7 +7,7 @@ The Periplus license does not replace these terms.
7
7
 
8
8
  ## Selectolax and Lexbor
9
9
 
10
- The backend pins Selectolax 0.4.11 (MIT) and its bundled Lexbor 3.1.0
10
+ The backend pins Selectolax 0.4.12 (MIT) and its bundled Lexbor 3.1.0
11
11
  (Apache-2.0). The DOM adapter's read-only native layouts follow Lexbor's DOM and
12
12
  HTML interface headers. Selectolax retains its packaged MIT license; Periplus
13
13
  includes Lexbor's license and notice under `licensing/third-party/lexbor/`
@@ -2,8 +2,8 @@
2
2
  license = "AGPL-3.0-only"
3
3
  license-files = ["licensing/LICENSE", "licensing/NOTICE", "licensing/*.md"]
4
4
  name = "periplus-python-sdk"
5
- version = "0.9.0"
6
- description = "Read-only Python client for the public Periplus query API"
5
+ version = "0.12.1"
6
+ description = "Typed Periplus platform client with SQL and notebook integration"
7
7
  readme = "README.md"
8
8
  requires-python = ">=3.11"
9
9
  dependencies = [
@@ -0,0 +1,186 @@
1
+ Metadata-Version: 2.4
2
+ Name: periplus-python-sdk
3
+ Version: 0.12.1
4
+ Summary: Typed Periplus platform client with SQL and notebook integration
5
+ License-Expression: AGPL-3.0-only
6
+ Project-URL: Repository, https://github.com/elei-io/periplus
7
+ Project-URL: Issues, https://github.com/elei-io/periplus/issues
8
+ Requires-Python: >=3.11
9
+ Description-Content-Type: text/markdown
10
+ License-File: licensing/LICENSE
11
+ License-File: licensing/NOTICE
12
+ License-File: licensing/README.md
13
+ License-File: licensing/THIRD_PARTY_NOTICES.md
14
+ Requires-Dist: httpx>=0.28
15
+ Requires-Dist: pydantic<3,>=2.12
16
+ Requires-Dist: sqlalchemy<3,>=2.0
17
+ Provides-Extra: notebook
18
+ Requires-Dist: marimo[sql]>=0.24.1; extra == "notebook"
19
+ Dynamic: license-file
20
+
21
+ # Periplus Python SDK
22
+
23
+ Typed organization operations and SQL access through the Periplus HTTP API.
24
+ The 0.12.0 platform interface requires the matching API release. It replaces the
25
+ discovery, retention and monitoring namespaces with captures, crawls, pins and
26
+ schedules, and includes the source-snapshot metadata introduced in 0.10.0.
27
+ Install from PyPI:
28
+
29
+ ```sh
30
+ python -m pip install --upgrade periplus-python-sdk
31
+ ```
32
+
33
+ ```python
34
+ from periplus_sdk import Client
35
+
36
+ with Client("https://api.periplus.dev", api_key="ppl_…") as client:
37
+ result = client.execute(
38
+ "SELECT capture_id, url FROM public_v1.captures LIMIT ?", [10]
39
+ )
40
+ print(result.columns, result.rows)
41
+ ```
42
+
43
+ The default origin is `https://api.periplus.dev`. Use `PERIPLUS_API_URL` to override it and `PERIPLUS_API_KEY` to
44
+ supply your personal organization API key. A key with `organization:sql:exec`
45
+ is required. Database usernames/passwords and anonymous access are not supported.
46
+ Use HTTPS outside loopback development. Connect directly to the API origin,
47
+ not the public marketing site.
48
+
49
+ Version 0.9.0 requires personal API keys instead of database credentials. Create
50
+ a Developer key in the app's Access → SDK & API. It inherits your current access in
51
+ that organization; SQL requires your membership to have `organization:sql:exec`.
52
+ For local development, install `./clients/periplus-python-sdk` from the repository
53
+ root and connect to `http://localhost:8000`.
54
+
55
+ `AsyncClient` accepts the same options. `prepare` explains a SELECT; `execute`
56
+ returns typed columns/rows for read-only queries. `schema()` returns
57
+ visible tables, column types/descriptions and helper documentation. All SQL uses
58
+ `POST /api/v1/sql`; schema discovery uses `GET /api/v1/schema`. ClickHouse enforces
59
+ permissions. Buffered SQL reads retry HTTP 502, 503 and 504 up to twice with bounded
60
+ backoff starting in 0.12.1. Set `sql_retries=0` to disable this, or choose up to three retries. Each
61
+ attempt may observe a newer corpus snapshot. Mutations, transport failures and
62
+ streamed queries are not retried automatically.
63
+
64
+ Public HTML joins use `capture_id` and `node_index`; `document_id` identifies exact
65
+ raw bytes. Public shorthand uses the `public_v1` schema.
66
+
67
+ In 0.10.0, `result.source_snapshot` is a typed `SourceSnapshot` containing
68
+ `layout_id` (UUID) and `publication_epoch` (integer), or `None` when no build-bound
69
+ public corpus was read. Buffered, asynchronous, streamed and DB-API results share
70
+ this contract. Compare both fields, not just the epoch. The identity describes
71
+ the public corpus inputs, not any native staff/external inputs. It does not retain
72
+ the data or request historical reads. This replaces the old nullable integer field;
73
+ use this SDK version with the corresponding API release.
74
+
75
+ For notebook/SQLAlchemy integration:
76
+
77
+ ```python
78
+ from periplus_sdk import sql_api
79
+ from sqlalchemy import text
80
+
81
+ engine = sql_api.create_engine(base_url="http://localhost:8000", api_key="ppl_…")
82
+ with engine.connect() as connection:
83
+ print(connection.execute(text("SELECT url FROM public_v1.captures LIMIT 5")).all())
84
+ engine.dispose()
85
+ ```
86
+
87
+ Marimo discovers accessible tables, views, typed columns and comments through the
88
+ schema endpoint. Reflection does not execute SQL. One SQLAlchemy Inspector caches
89
+ its metadata; call `inspector.clear_cache()` to refresh it. Missing metadata raises
90
+ an error rather than presenting an apparently complete empty schema.
91
+
92
+ The DB-API connection advertises the ClickHouse dialect and converts native
93
+ nullable integer, decimal, date and datetime types. Nested types retain JSON wire
94
+ values. Writes are not exposed through the query API. There are no client
95
+ transactions; each statement is independent.
96
+ Streaming cursors expose incomplete/truncated results explicitly; configure
97
+ `allow_partial` only when partial results suit the application.
98
+
99
+ See [the public schema](../../docs/SCHEMA.md) and [query boundary](../../docs/QUERY.md).
100
+
101
+ ## Organization resources
102
+
103
+ A Developer key selects your organization and uses your live membership access. Agent keys authenticate only to the hosted MCP endpoint, not this SDK.
104
+ `Client` and `AsyncClient` expose the same namespaces:
105
+
106
+ | Namespace | Operations |
107
+ | --- | --- |
108
+ | `identity`, `availability` | `get` |
109
+ | `captures`, `crawls` | `create`, `get`, `list`, `iter`, `pages`, `cancel` |
110
+ | `pins` | `create`, `get`, `list`, `iter`, `captures`, `update`, `delete` |
111
+ | `schedules` | `create`, `get`, `list`, `iter`, `pages`, `captures`, `update`, `pause`, `resume`, `delete` |
112
+ | `saved_queries` | `create`, `get`, `list`, `iter`, `rename`, `delete` |
113
+ | `members` | `list`, `update_role`, `remove` |
114
+ | `invitations` | `list`, `create`, `cancel` |
115
+ | `api_keys` | `create`, `list`, `iter`, `revoke` |
116
+ | `usage` | `get` |
117
+ | `query_history` | `list`, `iter`, `get`, `summary` |
118
+ | `audit` | `list`, `iter` |
119
+
120
+ Choose the resource by what you know and how long you need it:
121
+
122
+ | Resource | Use it when |
123
+ | --- | --- |
124
+ | `captures` | You know the URLs (1–100) and need them in SQL now |
125
+ | `crawls` | You know a starting point, not the URLs |
126
+ | `pins` | You need exact captures for longer than seven days |
127
+ | `schedules` | You need the same pages again later |
128
+
129
+ ```python
130
+ from periplus_sdk import Client
131
+
132
+ with Client() as client:
133
+ capture = client.captures.create(urls=["https://example.com/pricing"])
134
+ capture = client.captures.get(capture, wait_seconds=30)
135
+ if capture.finished and capture.queryable:
136
+ for page in client.captures.pages(capture).items:
137
+ print(page.url, page.status, page.capture_id)
138
+
139
+ crawl = client.crawls.create(seeds=["https://example.com/docs/"], page_limit=200,
140
+ allowed_paths=["/docs/*"])
141
+
142
+ pin = client.pins.create(
143
+ name="Research sources", days=90,
144
+ sql="SELECT capture_id FROM captures WHERE domain(url) = ?",
145
+ parameters=["example.com"],
146
+ )
147
+ schedule = client.schedules.create(name="Pricing", every_days=7,
148
+ urls=["https://example.com/pricing"])
149
+ print(pin.captures, schedule.projected_pages_per_month)
150
+ ```
151
+
152
+ Every capture is kept for seven days after collection unless a pin keeps it. A
153
+ capture always fetches again and never returns an existing capture, so check
154
+ coverage with SQL first. `request_id` (a capture or crawl `id`) is what you
155
+ ordered; each page's `capture_id` is what the crawler produced and is the SQL key.
156
+ `get(..., wait_seconds=N)` returns as soon as the request is finished and queryable,
157
+ or after at most 30 seconds; the SDK never polls on its own.
158
+
159
+ Pins and schedules accept explicit IDs/URLs or one read-only SQL query with
160
+ parameters, and are active immediately. SQL runs once at creation and its exact
161
+ result is locked in; running the same query beforehand is only an estimate.
162
+ Schedule URLs need not be in the corpus. `pin_days` on a capture, crawl or schedule
163
+ pins each successful capture it produces and requires `organization:pins:write`.
164
+ `pins.update` changes `days` (moving every capture's expiry by the difference) or
165
+ the name; `schedules.update` changes `every_days` or the name. Both accept a fetched
166
+ object or an ID with `expected_version`; conflicts are never retried. Deletion needs
167
+ no version: deleting a pin removes future protection, not captures.
168
+
169
+ Paginated lists return `Page[T]` (`items`, `next_cursor`). Pass cursors unchanged
170
+ or use `iter()`; membership/invitation lists are bounded snapshots instead.
171
+ Offset-based lists can shift during concurrent changes. Query history, request
172
+ page feeds and scheduled capture feeds use opaque keyset cursors. UTC usage ranges have an exclusive end date, at most 93 days. Usage reports pages
173
+ captured (broken down by capture, crawl or schedule), challenge-resolution pages and
174
+ pinned capture-days.
175
+ Query history is best-effort and expires after 30 days; it is not a billing ledger.
176
+
177
+ `ApiError` includes HTTP status, code, optional request ID, validation fields and
178
+ Retry-After seconds. `TransportError` means the outcome of a write can be unknown.
179
+ Keep creation IDs to reconcile; never blindly retry. Key creation is one-time:
180
+ `created.secret.get_secret_value()` reveals the secret and must only be used for
181
+ secure storage. Its ordinary representation is masked. Lost secrets cannot be
182
+ recovered.
183
+
184
+ For async streaming, use `async with await client.stream(sql) as stream` followed
185
+ by `async for batch in stream`. Streams validate completion and close on early
186
+ exit, errors, and cancellation. No threads or background polling are introduced.
@@ -11,10 +11,14 @@ src/periplus_python_sdk.egg-info/entry_points.txt
11
11
  src/periplus_python_sdk.egg-info/requires.txt
12
12
  src/periplus_python_sdk.egg-info/top_level.txt
13
13
  src/periplus_sdk/__init__.py
14
+ src/periplus_sdk/async_resources.py
15
+ src/periplus_sdk/async_stream.py
14
16
  src/periplus_sdk/client.py
15
17
  src/periplus_sdk/dbapi.py
16
18
  src/periplus_sdk/errors.py
17
19
  src/periplus_sdk/py.typed
20
+ src/periplus_sdk/resources.py
21
+ src/periplus_sdk/resources_types.py
18
22
  src/periplus_sdk/sql_api.py
19
23
  src/periplus_sdk/sqlalchemy.py
20
24
  src/periplus_sdk/stream.py
@@ -22,5 +26,6 @@ src/periplus_sdk/types.py
22
26
  tests/test_client.py
23
27
  tests/test_dbapi.py
24
28
  tests/test_notebook.py
29
+ tests/test_platform.py
25
30
  tests/test_sql_api.py
26
31
  tests/test_stream.py
@@ -0,0 +1,13 @@
1
+ """Typed organization operations and read-only SQL through the Periplus API."""
2
+ from .dbapi import connect
3
+ from .client import AsyncClient, Client
4
+ from .errors import ApiError, ConfigurationError, PeriplusError, ResponseError, TransportError
5
+ from .types import Diagnostic, PreparedQuery, QueryHelper, QueryHelpers, QueryResult, SourceSnapshot
6
+ from .resources_types import (FollowRule, QueryCondition, Page, Request, RequestPage, Pin, PinCapture, Schedule, SchedulePage, Member,
7
+ Invitation, ApiKey, CreatedApiKey, Identity, Usage)
8
+
9
+ __all__ = ["connect", "AsyncClient", "Client", "ApiError", "ConfigurationError", "PeriplusError",
10
+ "ResponseError", "TransportError", "Diagnostic", "PreparedQuery", "QueryHelper",
11
+ "QueryHelpers", "QueryResult", "FollowRule", "QueryCondition", "Page", "Request", "RequestPage", "Pin", "PinCapture",
12
+ "Schedule", "SchedulePage", "Member", "Invitation",
13
+ "ApiKey", "CreatedApiKey", "Identity", "Usage", "SourceSnapshot"]