periplus-python-sdk 0.11.0__tar.gz → 0.13.0__tar.gz
This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
- {periplus_python_sdk-0.11.0 → periplus_python_sdk-0.13.0}/PKG-INFO +57 -21
- {periplus_python_sdk-0.11.0 → periplus_python_sdk-0.13.0}/README.md +56 -20
- {periplus_python_sdk-0.11.0 → periplus_python_sdk-0.13.0}/licensing/THIRD_PARTY_NOTICES.md +1 -1
- {periplus_python_sdk-0.11.0 → periplus_python_sdk-0.13.0}/pyproject.toml +1 -1
- {periplus_python_sdk-0.11.0 → periplus_python_sdk-0.13.0}/src/periplus_python_sdk.egg-info/PKG-INFO +57 -21
- {periplus_python_sdk-0.11.0 → periplus_python_sdk-0.13.0}/src/periplus_python_sdk.egg-info/SOURCES.txt +0 -1
- {periplus_python_sdk-0.11.0 → periplus_python_sdk-0.13.0}/src/periplus_sdk/__init__.py +4 -2
- periplus_python_sdk-0.13.0/src/periplus_sdk/async_resources.py +373 -0
- {periplus_python_sdk-0.11.0 → periplus_python_sdk-0.13.0}/src/periplus_sdk/client.py +69 -25
- periplus_python_sdk-0.13.0/src/periplus_sdk/resources.py +374 -0
- periplus_python_sdk-0.13.0/src/periplus_sdk/resources_types.py +348 -0
- {periplus_python_sdk-0.11.0 → periplus_python_sdk-0.13.0}/tests/test_client.py +51 -1
- periplus_python_sdk-0.13.0/tests/test_platform.py +240 -0
- periplus_python_sdk-0.11.0/src/periplus_sdk/async_resources.py +0 -298
- periplus_python_sdk-0.11.0/src/periplus_sdk/resources.py +0 -297
- periplus_python_sdk-0.11.0/src/periplus_sdk/resources_types.py +0 -264
- periplus_python_sdk-0.11.0/src/periplus_sdk/selection.py +0 -29
- periplus_python_sdk-0.11.0/tests/test_platform.py +0 -99
- {periplus_python_sdk-0.11.0 → periplus_python_sdk-0.13.0}/licensing/LICENSE +0 -0
- {periplus_python_sdk-0.11.0 → periplus_python_sdk-0.13.0}/licensing/NOTICE +0 -0
- {periplus_python_sdk-0.11.0 → periplus_python_sdk-0.13.0}/licensing/README.md +0 -0
- {periplus_python_sdk-0.11.0 → periplus_python_sdk-0.13.0}/setup.cfg +0 -0
- {periplus_python_sdk-0.11.0 → periplus_python_sdk-0.13.0}/src/periplus_python_sdk.egg-info/dependency_links.txt +0 -0
- {periplus_python_sdk-0.11.0 → periplus_python_sdk-0.13.0}/src/periplus_python_sdk.egg-info/entry_points.txt +0 -0
- {periplus_python_sdk-0.11.0 → periplus_python_sdk-0.13.0}/src/periplus_python_sdk.egg-info/requires.txt +0 -0
- {periplus_python_sdk-0.11.0 → periplus_python_sdk-0.13.0}/src/periplus_python_sdk.egg-info/top_level.txt +0 -0
- {periplus_python_sdk-0.11.0 → periplus_python_sdk-0.13.0}/src/periplus_sdk/async_stream.py +0 -0
- {periplus_python_sdk-0.11.0 → periplus_python_sdk-0.13.0}/src/periplus_sdk/dbapi.py +0 -0
- {periplus_python_sdk-0.11.0 → periplus_python_sdk-0.13.0}/src/periplus_sdk/errors.py +0 -0
- {periplus_python_sdk-0.11.0 → periplus_python_sdk-0.13.0}/src/periplus_sdk/py.typed +0 -0
- {periplus_python_sdk-0.11.0 → periplus_python_sdk-0.13.0}/src/periplus_sdk/sql_api.py +0 -0
- {periplus_python_sdk-0.11.0 → periplus_python_sdk-0.13.0}/src/periplus_sdk/sqlalchemy.py +0 -0
- {periplus_python_sdk-0.11.0 → periplus_python_sdk-0.13.0}/src/periplus_sdk/stream.py +0 -0
- {periplus_python_sdk-0.11.0 → periplus_python_sdk-0.13.0}/src/periplus_sdk/types.py +0 -0
- {periplus_python_sdk-0.11.0 → periplus_python_sdk-0.13.0}/tests/test_dbapi.py +0 -0
- {periplus_python_sdk-0.11.0 → periplus_python_sdk-0.13.0}/tests/test_notebook.py +0 -0
- {periplus_python_sdk-0.11.0 → periplus_python_sdk-0.13.0}/tests/test_sql_api.py +0 -0
- {periplus_python_sdk-0.11.0 → periplus_python_sdk-0.13.0}/tests/test_stream.py +0 -0
|
@@ -1,6 +1,6 @@
|
|
|
1
1
|
Metadata-Version: 2.4
|
|
2
2
|
Name: periplus-python-sdk
|
|
3
|
-
Version: 0.
|
|
3
|
+
Version: 0.13.0
|
|
4
4
|
Summary: Typed Periplus platform client with SQL and notebook integration
|
|
5
5
|
License-Expression: AGPL-3.0-only
|
|
6
6
|
Project-URL: Repository, https://github.com/elei-io/periplus
|
|
@@ -21,8 +21,9 @@ Dynamic: license-file
|
|
|
21
21
|
# Periplus Python SDK
|
|
22
22
|
|
|
23
23
|
Typed organization operations and SQL access through the Periplus HTTP API.
|
|
24
|
-
The 0.
|
|
25
|
-
|
|
24
|
+
The 0.12.0 platform interface requires the matching API release. It replaces the
|
|
25
|
+
discovery, retention and monitoring namespaces with captures, crawls, pins and
|
|
26
|
+
schedules, and includes the source-snapshot metadata introduced in 0.10.0.
|
|
26
27
|
Install from PyPI:
|
|
27
28
|
|
|
28
29
|
```sh
|
|
@@ -46,7 +47,7 @@ Use HTTPS outside loopback development. Connect directly to the API origin,
|
|
|
46
47
|
not the public marketing site.
|
|
47
48
|
|
|
48
49
|
Version 0.9.0 requires personal API keys instead of database credentials. Create
|
|
49
|
-
a key in the app's
|
|
50
|
+
a Developer key in the app's Access → SDK & API. It inherits your current access in
|
|
50
51
|
that organization; SQL requires your membership to have `organization:sql:exec`.
|
|
51
52
|
For local development, install `./clients/periplus-python-sdk` from the repository
|
|
52
53
|
root and connect to `http://localhost:8000`.
|
|
@@ -55,9 +56,12 @@ root and connect to `http://localhost:8000`.
|
|
|
55
56
|
returns typed columns/rows for read-only queries. `schema()` returns
|
|
56
57
|
visible tables, column types/descriptions and helper documentation. All SQL uses
|
|
57
58
|
`POST /api/v1/sql`; schema discovery uses `GET /api/v1/schema`. ClickHouse enforces
|
|
58
|
-
permissions.
|
|
59
|
+
permissions. Buffered SQL reads retry HTTP 502, 503 and 504 up to twice with bounded
|
|
60
|
+
backoff starting in 0.12.1. Set `sql_retries=0` to disable this, or choose up to three retries. Each
|
|
61
|
+
attempt may observe a newer corpus snapshot. Mutations, transport failures and
|
|
62
|
+
streamed queries are not retried automatically.
|
|
59
63
|
|
|
60
|
-
Public HTML joins use `
|
|
64
|
+
Public HTML joins use `capture_id` and `node_index`; `document_id` identifies exact
|
|
61
65
|
raw bytes. Public shorthand uses the `public_v1` schema.
|
|
62
66
|
|
|
63
67
|
In 0.10.0, `result.source_snapshot` is a typed `SourceSnapshot` containing
|
|
@@ -96,15 +100,16 @@ See [the public schema](../../docs/SCHEMA.md) and [query boundary](../../docs/QU
|
|
|
96
100
|
|
|
97
101
|
## Organization resources
|
|
98
102
|
|
|
99
|
-
|
|
103
|
+
A Developer key selects your organization and uses your live membership access. Agent keys authenticate only to the hosted MCP endpoint, not this SDK.
|
|
100
104
|
`Client` and `AsyncClient` expose the same namespaces:
|
|
101
105
|
|
|
102
106
|
| Namespace | Operations |
|
|
103
107
|
| --- | --- |
|
|
104
108
|
| `identity`, `availability` | `get` |
|
|
105
|
-
| `
|
|
106
|
-
| `
|
|
107
|
-
| `
|
|
109
|
+
| `captures`, `crawls` | `create`, `get`, `list`, `iter`, `pages`, `cancel` |
|
|
110
|
+
| `pins` | `create`, `get`, `list`, `iter`, `captures`, `update`, `delete` |
|
|
111
|
+
| `schedules` | `create`, `get`, `list`, `iter`, `pages`, `captures`, `update`, `pause`, `resume`, `delete` |
|
|
112
|
+
| `saved_queries` | `create`, `get`, `list`, `iter`, `update`, `rename`, `delete` |
|
|
108
113
|
| `members` | `list`, `update_role`, `remove` |
|
|
109
114
|
| `invitations` | `list`, `create`, `cancel` |
|
|
110
115
|
| `api_keys` | `create`, `list`, `iter`, `revoke` |
|
|
@@ -112,30 +117,61 @@ The same key selects your organization and inherits your live membership access.
|
|
|
112
117
|
| `query_history` | `list`, `iter`, `get`, `summary` |
|
|
113
118
|
| `audit` | `list`, `iter` |
|
|
114
119
|
|
|
120
|
+
Choose the resource by what you know and how long you need it:
|
|
121
|
+
|
|
122
|
+
| Resource | Use it when |
|
|
123
|
+
| --- | --- |
|
|
124
|
+
| `captures` | You know the URLs (1–100) and need them in SQL now |
|
|
125
|
+
| `crawls` | You know a starting point, not the URLs |
|
|
126
|
+
| `pins` | You need exact captures for longer than seven days |
|
|
127
|
+
| `schedules` | You need the same pages again later |
|
|
128
|
+
|
|
115
129
|
```python
|
|
116
130
|
from periplus_sdk import Client
|
|
117
131
|
|
|
118
132
|
with Client() as client:
|
|
119
|
-
|
|
133
|
+
capture = client.captures.create(urls=["https://example.com/pricing"])
|
|
134
|
+
capture = client.captures.get(capture, wait_seconds=30)
|
|
135
|
+
if capture.finished and capture.queryable:
|
|
136
|
+
for page in client.captures.pages(capture).items:
|
|
137
|
+
print(page.url, page.status, page.capture_id)
|
|
138
|
+
|
|
139
|
+
crawl = client.crawls.create(seeds=["https://example.com/docs/"], page_limit=200,
|
|
140
|
+
allowed_paths=["/docs/*"])
|
|
141
|
+
|
|
142
|
+
pin = client.pins.create(
|
|
120
143
|
name="Research sources", days=90,
|
|
121
144
|
sql="SELECT capture_id FROM captures WHERE domain(url) = ?",
|
|
122
145
|
parameters=["example.com"],
|
|
123
146
|
)
|
|
124
|
-
|
|
125
|
-
|
|
126
|
-
|
|
147
|
+
schedule = client.schedules.create(name="Pricing", every_days=7,
|
|
148
|
+
urls=["https://example.com/pricing"])
|
|
149
|
+
print(pin.captures, schedule.projected_pages_per_month)
|
|
127
150
|
```
|
|
128
151
|
|
|
129
|
-
|
|
130
|
-
|
|
131
|
-
|
|
132
|
-
|
|
133
|
-
`
|
|
152
|
+
Every capture is kept for seven days after collection unless a pin keeps it. A
|
|
153
|
+
capture always fetches again and never returns an existing capture, so check
|
|
154
|
+
coverage with SQL first. `request_id` (a capture or crawl `id`) is what you
|
|
155
|
+
ordered; each page's `capture_id` is what the crawler produced and is the SQL key.
|
|
156
|
+
`get(..., wait_seconds=N)` returns as soon as the request is finished and queryable,
|
|
157
|
+
or after at most 30 seconds; the SDK never polls on its own.
|
|
158
|
+
|
|
159
|
+
Pins and schedules accept explicit IDs/URLs or one read-only SQL query with
|
|
160
|
+
parameters, and are active immediately. SQL runs once at creation and its exact
|
|
161
|
+
result is locked in; running the same query beforehand is only an estimate.
|
|
162
|
+
Schedule URLs need not be in the corpus. `pin_days` on a capture, crawl or schedule
|
|
163
|
+
pins each successful capture it produces and requires `organization:pins:write`.
|
|
164
|
+
`pins.update` changes `days` (moving every capture's expiry by the difference) or
|
|
165
|
+
the name; `schedules.update` changes `every_days` or the name. Both accept a fetched
|
|
166
|
+
object or an ID with `expected_version`; conflicts are never retried. Deletion needs
|
|
167
|
+
no version: deleting a pin removes future protection, not captures.
|
|
134
168
|
|
|
135
169
|
Paginated lists return `Page[T]` (`items`, `next_cursor`). Pass cursors unchanged
|
|
136
170
|
or use `iter()`; membership/invitation lists are bounded snapshots instead.
|
|
137
|
-
Offset-based lists can shift during concurrent changes. Query history
|
|
138
|
-
use keyset cursors. UTC usage ranges have an exclusive end date, at most 93 days.
|
|
171
|
+
Offset-based lists can shift during concurrent changes. Query history, request
|
|
172
|
+
page feeds and scheduled capture feeds use opaque keyset cursors. UTC usage ranges have an exclusive end date, at most 93 days. Usage reports pages
|
|
173
|
+
captured (broken down by capture, crawl or schedule), challenge-resolution pages and
|
|
174
|
+
pinned capture-days.
|
|
139
175
|
Query history is best-effort and expires after 30 days; it is not a billing ledger.
|
|
140
176
|
|
|
141
177
|
`ApiError` includes HTTP status, code, optional request ID, validation fields and
|
|
@@ -1,8 +1,9 @@
|
|
|
1
1
|
# Periplus Python SDK
|
|
2
2
|
|
|
3
3
|
Typed organization operations and SQL access through the Periplus HTTP API.
|
|
4
|
-
The 0.
|
|
5
|
-
|
|
4
|
+
The 0.12.0 platform interface requires the matching API release. It replaces the
|
|
5
|
+
discovery, retention and monitoring namespaces with captures, crawls, pins and
|
|
6
|
+
schedules, and includes the source-snapshot metadata introduced in 0.10.0.
|
|
6
7
|
Install from PyPI:
|
|
7
8
|
|
|
8
9
|
```sh
|
|
@@ -26,7 +27,7 @@ Use HTTPS outside loopback development. Connect directly to the API origin,
|
|
|
26
27
|
not the public marketing site.
|
|
27
28
|
|
|
28
29
|
Version 0.9.0 requires personal API keys instead of database credentials. Create
|
|
29
|
-
a key in the app's
|
|
30
|
+
a Developer key in the app's Access → SDK & API. It inherits your current access in
|
|
30
31
|
that organization; SQL requires your membership to have `organization:sql:exec`.
|
|
31
32
|
For local development, install `./clients/periplus-python-sdk` from the repository
|
|
32
33
|
root and connect to `http://localhost:8000`.
|
|
@@ -35,9 +36,12 @@ root and connect to `http://localhost:8000`.
|
|
|
35
36
|
returns typed columns/rows for read-only queries. `schema()` returns
|
|
36
37
|
visible tables, column types/descriptions and helper documentation. All SQL uses
|
|
37
38
|
`POST /api/v1/sql`; schema discovery uses `GET /api/v1/schema`. ClickHouse enforces
|
|
38
|
-
permissions.
|
|
39
|
+
permissions. Buffered SQL reads retry HTTP 502, 503 and 504 up to twice with bounded
|
|
40
|
+
backoff starting in 0.12.1. Set `sql_retries=0` to disable this, or choose up to three retries. Each
|
|
41
|
+
attempt may observe a newer corpus snapshot. Mutations, transport failures and
|
|
42
|
+
streamed queries are not retried automatically.
|
|
39
43
|
|
|
40
|
-
Public HTML joins use `
|
|
44
|
+
Public HTML joins use `capture_id` and `node_index`; `document_id` identifies exact
|
|
41
45
|
raw bytes. Public shorthand uses the `public_v1` schema.
|
|
42
46
|
|
|
43
47
|
In 0.10.0, `result.source_snapshot` is a typed `SourceSnapshot` containing
|
|
@@ -76,15 +80,16 @@ See [the public schema](../../docs/SCHEMA.md) and [query boundary](../../docs/QU
|
|
|
76
80
|
|
|
77
81
|
## Organization resources
|
|
78
82
|
|
|
79
|
-
|
|
83
|
+
A Developer key selects your organization and uses your live membership access. Agent keys authenticate only to the hosted MCP endpoint, not this SDK.
|
|
80
84
|
`Client` and `AsyncClient` expose the same namespaces:
|
|
81
85
|
|
|
82
86
|
| Namespace | Operations |
|
|
83
87
|
| --- | --- |
|
|
84
88
|
| `identity`, `availability` | `get` |
|
|
85
|
-
| `
|
|
86
|
-
| `
|
|
87
|
-
| `
|
|
89
|
+
| `captures`, `crawls` | `create`, `get`, `list`, `iter`, `pages`, `cancel` |
|
|
90
|
+
| `pins` | `create`, `get`, `list`, `iter`, `captures`, `update`, `delete` |
|
|
91
|
+
| `schedules` | `create`, `get`, `list`, `iter`, `pages`, `captures`, `update`, `pause`, `resume`, `delete` |
|
|
92
|
+
| `saved_queries` | `create`, `get`, `list`, `iter`, `update`, `rename`, `delete` |
|
|
88
93
|
| `members` | `list`, `update_role`, `remove` |
|
|
89
94
|
| `invitations` | `list`, `create`, `cancel` |
|
|
90
95
|
| `api_keys` | `create`, `list`, `iter`, `revoke` |
|
|
@@ -92,30 +97,61 @@ The same key selects your organization and inherits your live membership access.
|
|
|
92
97
|
| `query_history` | `list`, `iter`, `get`, `summary` |
|
|
93
98
|
| `audit` | `list`, `iter` |
|
|
94
99
|
|
|
100
|
+
Choose the resource by what you know and how long you need it:
|
|
101
|
+
|
|
102
|
+
| Resource | Use it when |
|
|
103
|
+
| --- | --- |
|
|
104
|
+
| `captures` | You know the URLs (1–100) and need them in SQL now |
|
|
105
|
+
| `crawls` | You know a starting point, not the URLs |
|
|
106
|
+
| `pins` | You need exact captures for longer than seven days |
|
|
107
|
+
| `schedules` | You need the same pages again later |
|
|
108
|
+
|
|
95
109
|
```python
|
|
96
110
|
from periplus_sdk import Client
|
|
97
111
|
|
|
98
112
|
with Client() as client:
|
|
99
|
-
|
|
113
|
+
capture = client.captures.create(urls=["https://example.com/pricing"])
|
|
114
|
+
capture = client.captures.get(capture, wait_seconds=30)
|
|
115
|
+
if capture.finished and capture.queryable:
|
|
116
|
+
for page in client.captures.pages(capture).items:
|
|
117
|
+
print(page.url, page.status, page.capture_id)
|
|
118
|
+
|
|
119
|
+
crawl = client.crawls.create(seeds=["https://example.com/docs/"], page_limit=200,
|
|
120
|
+
allowed_paths=["/docs/*"])
|
|
121
|
+
|
|
122
|
+
pin = client.pins.create(
|
|
100
123
|
name="Research sources", days=90,
|
|
101
124
|
sql="SELECT capture_id FROM captures WHERE domain(url) = ?",
|
|
102
125
|
parameters=["example.com"],
|
|
103
126
|
)
|
|
104
|
-
|
|
105
|
-
|
|
106
|
-
|
|
127
|
+
schedule = client.schedules.create(name="Pricing", every_days=7,
|
|
128
|
+
urls=["https://example.com/pricing"])
|
|
129
|
+
print(pin.captures, schedule.projected_pages_per_month)
|
|
107
130
|
```
|
|
108
131
|
|
|
109
|
-
|
|
110
|
-
|
|
111
|
-
|
|
112
|
-
|
|
113
|
-
`
|
|
132
|
+
Every capture is kept for seven days after collection unless a pin keeps it. A
|
|
133
|
+
capture always fetches again and never returns an existing capture, so check
|
|
134
|
+
coverage with SQL first. `request_id` (a capture or crawl `id`) is what you
|
|
135
|
+
ordered; each page's `capture_id` is what the crawler produced and is the SQL key.
|
|
136
|
+
`get(..., wait_seconds=N)` returns as soon as the request is finished and queryable,
|
|
137
|
+
or after at most 30 seconds; the SDK never polls on its own.
|
|
138
|
+
|
|
139
|
+
Pins and schedules accept explicit IDs/URLs or one read-only SQL query with
|
|
140
|
+
parameters, and are active immediately. SQL runs once at creation and its exact
|
|
141
|
+
result is locked in; running the same query beforehand is only an estimate.
|
|
142
|
+
Schedule URLs need not be in the corpus. `pin_days` on a capture, crawl or schedule
|
|
143
|
+
pins each successful capture it produces and requires `organization:pins:write`.
|
|
144
|
+
`pins.update` changes `days` (moving every capture's expiry by the difference) or
|
|
145
|
+
the name; `schedules.update` changes `every_days` or the name. Both accept a fetched
|
|
146
|
+
object or an ID with `expected_version`; conflicts are never retried. Deletion needs
|
|
147
|
+
no version: deleting a pin removes future protection, not captures.
|
|
114
148
|
|
|
115
149
|
Paginated lists return `Page[T]` (`items`, `next_cursor`). Pass cursors unchanged
|
|
116
150
|
or use `iter()`; membership/invitation lists are bounded snapshots instead.
|
|
117
|
-
Offset-based lists can shift during concurrent changes. Query history
|
|
118
|
-
use keyset cursors. UTC usage ranges have an exclusive end date, at most 93 days.
|
|
151
|
+
Offset-based lists can shift during concurrent changes. Query history, request
|
|
152
|
+
page feeds and scheduled capture feeds use opaque keyset cursors. UTC usage ranges have an exclusive end date, at most 93 days. Usage reports pages
|
|
153
|
+
captured (broken down by capture, crawl or schedule), challenge-resolution pages and
|
|
154
|
+
pinned capture-days.
|
|
119
155
|
Query history is best-effort and expires after 30 days; it is not a billing ledger.
|
|
120
156
|
|
|
121
157
|
`ApiError` includes HTTP status, code, optional request ID, validation fields and
|
|
@@ -7,7 +7,7 @@ The Periplus license does not replace these terms.
|
|
|
7
7
|
|
|
8
8
|
## Selectolax and Lexbor
|
|
9
9
|
|
|
10
|
-
The backend pins Selectolax 0.4.
|
|
10
|
+
The backend pins Selectolax 0.4.12 (MIT) and its bundled Lexbor 3.1.0
|
|
11
11
|
(Apache-2.0). The DOM adapter's read-only native layouts follow Lexbor's DOM and
|
|
12
12
|
HTML interface headers. Selectolax retains its packaged MIT license; Periplus
|
|
13
13
|
includes Lexbor's license and notice under `licensing/third-party/lexbor/`
|
|
@@ -2,7 +2,7 @@
|
|
|
2
2
|
license = "AGPL-3.0-only"
|
|
3
3
|
license-files = ["licensing/LICENSE", "licensing/NOTICE", "licensing/*.md"]
|
|
4
4
|
name = "periplus-python-sdk"
|
|
5
|
-
version = "0.
|
|
5
|
+
version = "0.13.0"
|
|
6
6
|
description = "Typed Periplus platform client with SQL and notebook integration"
|
|
7
7
|
readme = "README.md"
|
|
8
8
|
requires-python = ">=3.11"
|
{periplus_python_sdk-0.11.0 → periplus_python_sdk-0.13.0}/src/periplus_python_sdk.egg-info/PKG-INFO
RENAMED
|
@@ -1,6 +1,6 @@
|
|
|
1
1
|
Metadata-Version: 2.4
|
|
2
2
|
Name: periplus-python-sdk
|
|
3
|
-
Version: 0.
|
|
3
|
+
Version: 0.13.0
|
|
4
4
|
Summary: Typed Periplus platform client with SQL and notebook integration
|
|
5
5
|
License-Expression: AGPL-3.0-only
|
|
6
6
|
Project-URL: Repository, https://github.com/elei-io/periplus
|
|
@@ -21,8 +21,9 @@ Dynamic: license-file
|
|
|
21
21
|
# Periplus Python SDK
|
|
22
22
|
|
|
23
23
|
Typed organization operations and SQL access through the Periplus HTTP API.
|
|
24
|
-
The 0.
|
|
25
|
-
|
|
24
|
+
The 0.12.0 platform interface requires the matching API release. It replaces the
|
|
25
|
+
discovery, retention and monitoring namespaces with captures, crawls, pins and
|
|
26
|
+
schedules, and includes the source-snapshot metadata introduced in 0.10.0.
|
|
26
27
|
Install from PyPI:
|
|
27
28
|
|
|
28
29
|
```sh
|
|
@@ -46,7 +47,7 @@ Use HTTPS outside loopback development. Connect directly to the API origin,
|
|
|
46
47
|
not the public marketing site.
|
|
47
48
|
|
|
48
49
|
Version 0.9.0 requires personal API keys instead of database credentials. Create
|
|
49
|
-
a key in the app's
|
|
50
|
+
a Developer key in the app's Access → SDK & API. It inherits your current access in
|
|
50
51
|
that organization; SQL requires your membership to have `organization:sql:exec`.
|
|
51
52
|
For local development, install `./clients/periplus-python-sdk` from the repository
|
|
52
53
|
root and connect to `http://localhost:8000`.
|
|
@@ -55,9 +56,12 @@ root and connect to `http://localhost:8000`.
|
|
|
55
56
|
returns typed columns/rows for read-only queries. `schema()` returns
|
|
56
57
|
visible tables, column types/descriptions and helper documentation. All SQL uses
|
|
57
58
|
`POST /api/v1/sql`; schema discovery uses `GET /api/v1/schema`. ClickHouse enforces
|
|
58
|
-
permissions.
|
|
59
|
+
permissions. Buffered SQL reads retry HTTP 502, 503 and 504 up to twice with bounded
|
|
60
|
+
backoff starting in 0.12.1. Set `sql_retries=0` to disable this, or choose up to three retries. Each
|
|
61
|
+
attempt may observe a newer corpus snapshot. Mutations, transport failures and
|
|
62
|
+
streamed queries are not retried automatically.
|
|
59
63
|
|
|
60
|
-
Public HTML joins use `
|
|
64
|
+
Public HTML joins use `capture_id` and `node_index`; `document_id` identifies exact
|
|
61
65
|
raw bytes. Public shorthand uses the `public_v1` schema.
|
|
62
66
|
|
|
63
67
|
In 0.10.0, `result.source_snapshot` is a typed `SourceSnapshot` containing
|
|
@@ -96,15 +100,16 @@ See [the public schema](../../docs/SCHEMA.md) and [query boundary](../../docs/QU
|
|
|
96
100
|
|
|
97
101
|
## Organization resources
|
|
98
102
|
|
|
99
|
-
|
|
103
|
+
A Developer key selects your organization and uses your live membership access. Agent keys authenticate only to the hosted MCP endpoint, not this SDK.
|
|
100
104
|
`Client` and `AsyncClient` expose the same namespaces:
|
|
101
105
|
|
|
102
106
|
| Namespace | Operations |
|
|
103
107
|
| --- | --- |
|
|
104
108
|
| `identity`, `availability` | `get` |
|
|
105
|
-
| `
|
|
106
|
-
| `
|
|
107
|
-
| `
|
|
109
|
+
| `captures`, `crawls` | `create`, `get`, `list`, `iter`, `pages`, `cancel` |
|
|
110
|
+
| `pins` | `create`, `get`, `list`, `iter`, `captures`, `update`, `delete` |
|
|
111
|
+
| `schedules` | `create`, `get`, `list`, `iter`, `pages`, `captures`, `update`, `pause`, `resume`, `delete` |
|
|
112
|
+
| `saved_queries` | `create`, `get`, `list`, `iter`, `update`, `rename`, `delete` |
|
|
108
113
|
| `members` | `list`, `update_role`, `remove` |
|
|
109
114
|
| `invitations` | `list`, `create`, `cancel` |
|
|
110
115
|
| `api_keys` | `create`, `list`, `iter`, `revoke` |
|
|
@@ -112,30 +117,61 @@ The same key selects your organization and inherits your live membership access.
|
|
|
112
117
|
| `query_history` | `list`, `iter`, `get`, `summary` |
|
|
113
118
|
| `audit` | `list`, `iter` |
|
|
114
119
|
|
|
120
|
+
Choose the resource by what you know and how long you need it:
|
|
121
|
+
|
|
122
|
+
| Resource | Use it when |
|
|
123
|
+
| --- | --- |
|
|
124
|
+
| `captures` | You know the URLs (1–100) and need them in SQL now |
|
|
125
|
+
| `crawls` | You know a starting point, not the URLs |
|
|
126
|
+
| `pins` | You need exact captures for longer than seven days |
|
|
127
|
+
| `schedules` | You need the same pages again later |
|
|
128
|
+
|
|
115
129
|
```python
|
|
116
130
|
from periplus_sdk import Client
|
|
117
131
|
|
|
118
132
|
with Client() as client:
|
|
119
|
-
|
|
133
|
+
capture = client.captures.create(urls=["https://example.com/pricing"])
|
|
134
|
+
capture = client.captures.get(capture, wait_seconds=30)
|
|
135
|
+
if capture.finished and capture.queryable:
|
|
136
|
+
for page in client.captures.pages(capture).items:
|
|
137
|
+
print(page.url, page.status, page.capture_id)
|
|
138
|
+
|
|
139
|
+
crawl = client.crawls.create(seeds=["https://example.com/docs/"], page_limit=200,
|
|
140
|
+
allowed_paths=["/docs/*"])
|
|
141
|
+
|
|
142
|
+
pin = client.pins.create(
|
|
120
143
|
name="Research sources", days=90,
|
|
121
144
|
sql="SELECT capture_id FROM captures WHERE domain(url) = ?",
|
|
122
145
|
parameters=["example.com"],
|
|
123
146
|
)
|
|
124
|
-
|
|
125
|
-
|
|
126
|
-
|
|
147
|
+
schedule = client.schedules.create(name="Pricing", every_days=7,
|
|
148
|
+
urls=["https://example.com/pricing"])
|
|
149
|
+
print(pin.captures, schedule.projected_pages_per_month)
|
|
127
150
|
```
|
|
128
151
|
|
|
129
|
-
|
|
130
|
-
|
|
131
|
-
|
|
132
|
-
|
|
133
|
-
`
|
|
152
|
+
Every capture is kept for seven days after collection unless a pin keeps it. A
|
|
153
|
+
capture always fetches again and never returns an existing capture, so check
|
|
154
|
+
coverage with SQL first. `request_id` (a capture or crawl `id`) is what you
|
|
155
|
+
ordered; each page's `capture_id` is what the crawler produced and is the SQL key.
|
|
156
|
+
`get(..., wait_seconds=N)` returns as soon as the request is finished and queryable,
|
|
157
|
+
or after at most 30 seconds; the SDK never polls on its own.
|
|
158
|
+
|
|
159
|
+
Pins and schedules accept explicit IDs/URLs or one read-only SQL query with
|
|
160
|
+
parameters, and are active immediately. SQL runs once at creation and its exact
|
|
161
|
+
result is locked in; running the same query beforehand is only an estimate.
|
|
162
|
+
Schedule URLs need not be in the corpus. `pin_days` on a capture, crawl or schedule
|
|
163
|
+
pins each successful capture it produces and requires `organization:pins:write`.
|
|
164
|
+
`pins.update` changes `days` (moving every capture's expiry by the difference) or
|
|
165
|
+
the name; `schedules.update` changes `every_days` or the name. Both accept a fetched
|
|
166
|
+
object or an ID with `expected_version`; conflicts are never retried. Deletion needs
|
|
167
|
+
no version: deleting a pin removes future protection, not captures.
|
|
134
168
|
|
|
135
169
|
Paginated lists return `Page[T]` (`items`, `next_cursor`). Pass cursors unchanged
|
|
136
170
|
or use `iter()`; membership/invitation lists are bounded snapshots instead.
|
|
137
|
-
Offset-based lists can shift during concurrent changes. Query history
|
|
138
|
-
use keyset cursors. UTC usage ranges have an exclusive end date, at most 93 days.
|
|
171
|
+
Offset-based lists can shift during concurrent changes. Query history, request
|
|
172
|
+
page feeds and scheduled capture feeds use opaque keyset cursors. UTC usage ranges have an exclusive end date, at most 93 days. Usage reports pages
|
|
173
|
+
captured (broken down by capture, crawl or schedule), challenge-resolution pages and
|
|
174
|
+
pinned capture-days.
|
|
139
175
|
Query history is best-effort and expires after 30 days; it is not a billing ledger.
|
|
140
176
|
|
|
141
177
|
`ApiError` includes HTTP status, code, optional request ID, validation fields and
|
|
@@ -3,9 +3,11 @@ from .dbapi import connect
|
|
|
3
3
|
from .client import AsyncClient, Client
|
|
4
4
|
from .errors import ApiError, ConfigurationError, PeriplusError, ResponseError, TransportError
|
|
5
5
|
from .types import Diagnostic, PreparedQuery, QueryHelper, QueryHelpers, QueryResult, SourceSnapshot
|
|
6
|
-
from .resources_types import Page,
|
|
6
|
+
from .resources_types import (FollowRule, QueryCondition, Page, Request, RequestPage, Pin, PinCapture, Schedule, SchedulePage, Member,
|
|
7
|
+
Invitation, ApiKey, CreatedApiKey, Identity, Usage)
|
|
7
8
|
|
|
8
9
|
__all__ = ["connect", "AsyncClient", "Client", "ApiError", "ConfigurationError", "PeriplusError",
|
|
9
10
|
"ResponseError", "TransportError", "Diagnostic", "PreparedQuery", "QueryHelper",
|
|
10
|
-
"QueryHelpers", "QueryResult", "Page", "
|
|
11
|
+
"QueryHelpers", "QueryResult", "FollowRule", "QueryCondition", "Page", "Request", "RequestPage", "Pin", "PinCapture",
|
|
12
|
+
"Schedule", "SchedulePage", "Member", "Invitation",
|
|
11
13
|
"ApiKey", "CreatedApiKey", "Identity", "Usage", "SourceSnapshot"]
|