periplus-python-sdk 0.9.0__py3-none-any.whl → 0.12.1__py3-none-any.whl
This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
- periplus_python_sdk-0.12.1.dist-info/METADATA +186 -0
- periplus_python_sdk-0.12.1.dist-info/RECORD +22 -0
- {periplus_python_sdk-0.9.0.dist-info → periplus_python_sdk-0.12.1.dist-info}/licenses/licensing/THIRD_PARTY_NOTICES.md +1 -1
- periplus_sdk/__init__.py +7 -3
- periplus_sdk/async_resources.py +366 -0
- periplus_sdk/async_stream.py +104 -0
- periplus_sdk/client.py +166 -13
- periplus_sdk/errors.py +3 -1
- periplus_sdk/resources.py +367 -0
- periplus_sdk/resources_types.py +348 -0
- periplus_sdk/sqlalchemy.py +25 -25
- periplus_sdk/stream.py +8 -2
- periplus_sdk/types.py +91 -3
- periplus_python_sdk-0.9.0.dist-info/METADATA +0 -83
- periplus_python_sdk-0.9.0.dist-info/RECORD +0 -18
- {periplus_python_sdk-0.9.0.dist-info → periplus_python_sdk-0.12.1.dist-info}/WHEEL +0 -0
- {periplus_python_sdk-0.9.0.dist-info → periplus_python_sdk-0.12.1.dist-info}/entry_points.txt +0 -0
- {periplus_python_sdk-0.9.0.dist-info → periplus_python_sdk-0.12.1.dist-info}/licenses/licensing/LICENSE +0 -0
- {periplus_python_sdk-0.9.0.dist-info → periplus_python_sdk-0.12.1.dist-info}/licenses/licensing/NOTICE +0 -0
- {periplus_python_sdk-0.9.0.dist-info → periplus_python_sdk-0.12.1.dist-info}/licenses/licensing/README.md +0 -0
- {periplus_python_sdk-0.9.0.dist-info → periplus_python_sdk-0.12.1.dist-info}/top_level.txt +0 -0
|
@@ -0,0 +1,186 @@
|
|
|
1
|
+
Metadata-Version: 2.4
|
|
2
|
+
Name: periplus-python-sdk
|
|
3
|
+
Version: 0.12.1
|
|
4
|
+
Summary: Typed Periplus platform client with SQL and notebook integration
|
|
5
|
+
License-Expression: AGPL-3.0-only
|
|
6
|
+
Project-URL: Repository, https://github.com/elei-io/periplus
|
|
7
|
+
Project-URL: Issues, https://github.com/elei-io/periplus/issues
|
|
8
|
+
Requires-Python: >=3.11
|
|
9
|
+
Description-Content-Type: text/markdown
|
|
10
|
+
License-File: licensing/LICENSE
|
|
11
|
+
License-File: licensing/NOTICE
|
|
12
|
+
License-File: licensing/README.md
|
|
13
|
+
License-File: licensing/THIRD_PARTY_NOTICES.md
|
|
14
|
+
Requires-Dist: httpx>=0.28
|
|
15
|
+
Requires-Dist: pydantic<3,>=2.12
|
|
16
|
+
Requires-Dist: sqlalchemy<3,>=2.0
|
|
17
|
+
Provides-Extra: notebook
|
|
18
|
+
Requires-Dist: marimo[sql]>=0.24.1; extra == "notebook"
|
|
19
|
+
Dynamic: license-file
|
|
20
|
+
|
|
21
|
+
# Periplus Python SDK
|
|
22
|
+
|
|
23
|
+
Typed organization operations and SQL access through the Periplus HTTP API.
|
|
24
|
+
The 0.12.0 platform interface requires the matching API release. It replaces the
|
|
25
|
+
discovery, retention and monitoring namespaces with captures, crawls, pins and
|
|
26
|
+
schedules, and includes the source-snapshot metadata introduced in 0.10.0.
|
|
27
|
+
Install from PyPI:
|
|
28
|
+
|
|
29
|
+
```sh
|
|
30
|
+
python -m pip install --upgrade periplus-python-sdk
|
|
31
|
+
```
|
|
32
|
+
|
|
33
|
+
```python
|
|
34
|
+
from periplus_sdk import Client
|
|
35
|
+
|
|
36
|
+
with Client("https://api.periplus.dev", api_key="ppl_…") as client:
|
|
37
|
+
result = client.execute(
|
|
38
|
+
"SELECT capture_id, url FROM public_v1.captures LIMIT ?", [10]
|
|
39
|
+
)
|
|
40
|
+
print(result.columns, result.rows)
|
|
41
|
+
```
|
|
42
|
+
|
|
43
|
+
The default origin is `https://api.periplus.dev`. Use `PERIPLUS_API_URL` to override it and `PERIPLUS_API_KEY` to
|
|
44
|
+
supply your personal organization API key. A key with `organization:sql:exec`
|
|
45
|
+
is required. Database usernames/passwords and anonymous access are not supported.
|
|
46
|
+
Use HTTPS outside loopback development. Connect directly to the API origin,
|
|
47
|
+
not the public marketing site.
|
|
48
|
+
|
|
49
|
+
Version 0.9.0 requires personal API keys instead of database credentials. Create
|
|
50
|
+
a Developer key in the app's Access → SDK & API. It inherits your current access in
|
|
51
|
+
that organization; SQL requires your membership to have `organization:sql:exec`.
|
|
52
|
+
For local development, install `./clients/periplus-python-sdk` from the repository
|
|
53
|
+
root and connect to `http://localhost:8000`.
|
|
54
|
+
|
|
55
|
+
`AsyncClient` accepts the same options. `prepare` explains a SELECT; `execute`
|
|
56
|
+
returns typed columns/rows for read-only queries. `schema()` returns
|
|
57
|
+
visible tables, column types/descriptions and helper documentation. All SQL uses
|
|
58
|
+
`POST /api/v1/sql`; schema discovery uses `GET /api/v1/schema`. ClickHouse enforces
|
|
59
|
+
permissions. Buffered SQL reads retry HTTP 502, 503 and 504 up to twice with bounded
|
|
60
|
+
backoff starting in 0.12.1. Set `sql_retries=0` to disable this, or choose up to three retries. Each
|
|
61
|
+
attempt may observe a newer corpus snapshot. Mutations, transport failures and
|
|
62
|
+
streamed queries are not retried automatically.
|
|
63
|
+
|
|
64
|
+
Public HTML joins use `capture_id` and `node_index`; `document_id` identifies exact
|
|
65
|
+
raw bytes. Public shorthand uses the `public_v1` schema.
|
|
66
|
+
|
|
67
|
+
In 0.10.0, `result.source_snapshot` is a typed `SourceSnapshot` containing
|
|
68
|
+
`layout_id` (UUID) and `publication_epoch` (integer), or `None` when no build-bound
|
|
69
|
+
public corpus was read. Buffered, asynchronous, streamed and DB-API results share
|
|
70
|
+
this contract. Compare both fields, not just the epoch. The identity describes
|
|
71
|
+
the public corpus inputs, not any native staff/external inputs. It does not retain
|
|
72
|
+
the data or request historical reads. This replaces the old nullable integer field;
|
|
73
|
+
use this SDK version with the corresponding API release.
|
|
74
|
+
|
|
75
|
+
For notebook/SQLAlchemy integration:
|
|
76
|
+
|
|
77
|
+
```python
|
|
78
|
+
from periplus_sdk import sql_api
|
|
79
|
+
from sqlalchemy import text
|
|
80
|
+
|
|
81
|
+
engine = sql_api.create_engine(base_url="http://localhost:8000", api_key="ppl_…")
|
|
82
|
+
with engine.connect() as connection:
|
|
83
|
+
print(connection.execute(text("SELECT url FROM public_v1.captures LIMIT 5")).all())
|
|
84
|
+
engine.dispose()
|
|
85
|
+
```
|
|
86
|
+
|
|
87
|
+
Marimo discovers accessible tables, views, typed columns and comments through the
|
|
88
|
+
schema endpoint. Reflection does not execute SQL. One SQLAlchemy Inspector caches
|
|
89
|
+
its metadata; call `inspector.clear_cache()` to refresh it. Missing metadata raises
|
|
90
|
+
an error rather than presenting an apparently complete empty schema.
|
|
91
|
+
|
|
92
|
+
The DB-API connection advertises the ClickHouse dialect and converts native
|
|
93
|
+
nullable integer, decimal, date and datetime types. Nested types retain JSON wire
|
|
94
|
+
values. Writes are not exposed through the query API. There are no client
|
|
95
|
+
transactions; each statement is independent.
|
|
96
|
+
Streaming cursors expose incomplete/truncated results explicitly; configure
|
|
97
|
+
`allow_partial` only when partial results suit the application.
|
|
98
|
+
|
|
99
|
+
See [the public schema](../../docs/SCHEMA.md) and [query boundary](../../docs/QUERY.md).
|
|
100
|
+
|
|
101
|
+
## Organization resources
|
|
102
|
+
|
|
103
|
+
A Developer key selects your organization and uses your live membership access. Agent keys authenticate only to the hosted MCP endpoint, not this SDK.
|
|
104
|
+
`Client` and `AsyncClient` expose the same namespaces:
|
|
105
|
+
|
|
106
|
+
| Namespace | Operations |
|
|
107
|
+
| --- | --- |
|
|
108
|
+
| `identity`, `availability` | `get` |
|
|
109
|
+
| `captures`, `crawls` | `create`, `get`, `list`, `iter`, `pages`, `cancel` |
|
|
110
|
+
| `pins` | `create`, `get`, `list`, `iter`, `captures`, `update`, `delete` |
|
|
111
|
+
| `schedules` | `create`, `get`, `list`, `iter`, `pages`, `captures`, `update`, `pause`, `resume`, `delete` |
|
|
112
|
+
| `saved_queries` | `create`, `get`, `list`, `iter`, `rename`, `delete` |
|
|
113
|
+
| `members` | `list`, `update_role`, `remove` |
|
|
114
|
+
| `invitations` | `list`, `create`, `cancel` |
|
|
115
|
+
| `api_keys` | `create`, `list`, `iter`, `revoke` |
|
|
116
|
+
| `usage` | `get` |
|
|
117
|
+
| `query_history` | `list`, `iter`, `get`, `summary` |
|
|
118
|
+
| `audit` | `list`, `iter` |
|
|
119
|
+
|
|
120
|
+
Choose the resource by what you know and how long you need it:
|
|
121
|
+
|
|
122
|
+
| Resource | Use it when |
|
|
123
|
+
| --- | --- |
|
|
124
|
+
| `captures` | You know the URLs (1–100) and need them in SQL now |
|
|
125
|
+
| `crawls` | You know a starting point, not the URLs |
|
|
126
|
+
| `pins` | You need exact captures for longer than seven days |
|
|
127
|
+
| `schedules` | You need the same pages again later |
|
|
128
|
+
|
|
129
|
+
```python
|
|
130
|
+
from periplus_sdk import Client
|
|
131
|
+
|
|
132
|
+
with Client() as client:
|
|
133
|
+
capture = client.captures.create(urls=["https://example.com/pricing"])
|
|
134
|
+
capture = client.captures.get(capture, wait_seconds=30)
|
|
135
|
+
if capture.finished and capture.queryable:
|
|
136
|
+
for page in client.captures.pages(capture).items:
|
|
137
|
+
print(page.url, page.status, page.capture_id)
|
|
138
|
+
|
|
139
|
+
crawl = client.crawls.create(seeds=["https://example.com/docs/"], page_limit=200,
|
|
140
|
+
allowed_paths=["/docs/*"])
|
|
141
|
+
|
|
142
|
+
pin = client.pins.create(
|
|
143
|
+
name="Research sources", days=90,
|
|
144
|
+
sql="SELECT capture_id FROM captures WHERE domain(url) = ?",
|
|
145
|
+
parameters=["example.com"],
|
|
146
|
+
)
|
|
147
|
+
schedule = client.schedules.create(name="Pricing", every_days=7,
|
|
148
|
+
urls=["https://example.com/pricing"])
|
|
149
|
+
print(pin.captures, schedule.projected_pages_per_month)
|
|
150
|
+
```
|
|
151
|
+
|
|
152
|
+
Every capture is kept for seven days after collection unless a pin keeps it. A
|
|
153
|
+
capture always fetches again and never returns an existing capture, so check
|
|
154
|
+
coverage with SQL first. `request_id` (a capture or crawl `id`) is what you
|
|
155
|
+
ordered; each page's `capture_id` is what the crawler produced and is the SQL key.
|
|
156
|
+
`get(..., wait_seconds=N)` returns as soon as the request is finished and queryable,
|
|
157
|
+
or after at most 30 seconds; the SDK never polls on its own.
|
|
158
|
+
|
|
159
|
+
Pins and schedules accept explicit IDs/URLs or one read-only SQL query with
|
|
160
|
+
parameters, and are active immediately. SQL runs once at creation and its exact
|
|
161
|
+
result is locked in; running the same query beforehand is only an estimate.
|
|
162
|
+
Schedule URLs need not be in the corpus. `pin_days` on a capture, crawl or schedule
|
|
163
|
+
pins each successful capture it produces and requires `organization:pins:write`.
|
|
164
|
+
`pins.update` changes `days` (moving every capture's expiry by the difference) or
|
|
165
|
+
the name; `schedules.update` changes `every_days` or the name. Both accept a fetched
|
|
166
|
+
object or an ID with `expected_version`; conflicts are never retried. Deletion needs
|
|
167
|
+
no version: deleting a pin removes future protection, not captures.
|
|
168
|
+
|
|
169
|
+
Paginated lists return `Page[T]` (`items`, `next_cursor`). Pass cursors unchanged
|
|
170
|
+
or use `iter()`; membership/invitation lists are bounded snapshots instead.
|
|
171
|
+
Offset-based lists can shift during concurrent changes. Query history, request
|
|
172
|
+
page feeds and scheduled capture feeds use opaque keyset cursors. UTC usage ranges have an exclusive end date, at most 93 days. Usage reports pages
|
|
173
|
+
captured (broken down by capture, crawl or schedule), challenge-resolution pages and
|
|
174
|
+
pinned capture-days.
|
|
175
|
+
Query history is best-effort and expires after 30 days; it is not a billing ledger.
|
|
176
|
+
|
|
177
|
+
`ApiError` includes HTTP status, code, optional request ID, validation fields and
|
|
178
|
+
Retry-After seconds. `TransportError` means the outcome of a write can be unknown.
|
|
179
|
+
Keep creation IDs to reconcile; never blindly retry. Key creation is one-time:
|
|
180
|
+
`created.secret.get_secret_value()` reveals the secret and must only be used for
|
|
181
|
+
secure storage. Its ordinary representation is masked. Lost secrets cannot be
|
|
182
|
+
recovered.
|
|
183
|
+
|
|
184
|
+
For async streaming, use `async with await client.stream(sql) as stream` followed
|
|
185
|
+
by `async for batch in stream`. Streams validate completion and close on early
|
|
186
|
+
exit, errors, and cancellation. No threads or background polling are introduced.
|
|
@@ -0,0 +1,22 @@
|
|
|
1
|
+
periplus_python_sdk-0.12.1.dist-info/licenses/licensing/LICENSE,sha256=DZak_2itbUtvHzD3E7GNUYSRK6jdOJ-GqncQ2weavLA,34523
|
|
2
|
+
periplus_python_sdk-0.12.1.dist-info/licenses/licensing/NOTICE,sha256=W4W3pzZe-a0ebw4MxgWVmFGj7NSGL1qj8hFx4YG9iF0,105
|
|
3
|
+
periplus_python_sdk-0.12.1.dist-info/licenses/licensing/README.md,sha256=NABTC2wm2Y9iRNMnlMeyVMA2dsAw7XRmCrsV1B2TQdk,1996
|
|
4
|
+
periplus_python_sdk-0.12.1.dist-info/licenses/licensing/THIRD_PARTY_NOTICES.md,sha256=-egdv-dVsKhyRE2pPaHZgg4XYdpQfpH0VwM-sHYjUTg,9643
|
|
5
|
+
periplus_sdk/__init__.py,sha256=QzgQOvgN2YBTr6Vf65uCwKzAPjjnxsq4kdXy2XWxRAs,1020
|
|
6
|
+
periplus_sdk/async_resources.py,sha256=iSs3w-T6iCbQroWHU4aplKqVlnB7MCUlEf_na0FPym8,20073
|
|
7
|
+
periplus_sdk/async_stream.py,sha256=O5hO_w_-ooQuDV1fOZRX4Qei1-w7Wri8eIlsn3yCpGQ,4592
|
|
8
|
+
periplus_sdk/client.py,sha256=zVt1uso8LXQuwjVjOz3Nm9PH8bCAv1ZwLbuAvSj5BAI,15592
|
|
9
|
+
periplus_sdk/dbapi.py,sha256=dkJA-DkrMtcgMGjLQ5Vv8-pb6riRrAaG4Yqrn22uRZY,12064
|
|
10
|
+
periplus_sdk/errors.py,sha256=nnbmRzEu1GEFGeKNsLqEo6pHCFu0oFy__5t-yBLO2hw,965
|
|
11
|
+
periplus_sdk/py.typed,sha256=AbpHGcgLb-kRsJGnwFEktk7uzpZOCcBY74-YBdrKVGs,1
|
|
12
|
+
periplus_sdk/resources.py,sha256=JX7UEs4aHx_n2yErFdnDqaRHrsJtFSrWfSw1EaFm0vA,19680
|
|
13
|
+
periplus_sdk/resources_types.py,sha256=_UBaTWaut8xq9jDcrJKZdswTDFjSi179P7V6LFjOhQU,8949
|
|
14
|
+
periplus_sdk/sql_api.py,sha256=uUWdwMRSZFMNXXy3-yLozcuT6WluvT6BTjYDB0lNrAU,1917
|
|
15
|
+
periplus_sdk/sqlalchemy.py,sha256=XE_OzdDh9OZ8goWWRoKlBzp7Q0FM-eS9-hjH7Ofzy-E,6292
|
|
16
|
+
periplus_sdk/stream.py,sha256=6Ytb6FkP3Y5C4sDA_SDdU0MJHBQdgHde1lNM6YxHdIQ,5560
|
|
17
|
+
periplus_sdk/types.py,sha256=TSaQRjN5vng_yV4hjZV03s5sU_RHL874qJe40-Hb1KQ,3644
|
|
18
|
+
periplus_python_sdk-0.12.1.dist-info/METADATA,sha256=IbyLcCvef30Umg7LyHPed2lyKRzBPGKsnDTdiKLIN_U,9198
|
|
19
|
+
periplus_python_sdk-0.12.1.dist-info/WHEEL,sha256=YVMoNqKzERt-wjUZwJ33xBGAwnFl-4cqbYkTtWa4itE,91
|
|
20
|
+
periplus_python_sdk-0.12.1.dist-info/entry_points.txt,sha256=Pr14L_7AhLinq-4qDxB4awFVubrvR1BEfkaFVRpdCbU,73
|
|
21
|
+
periplus_python_sdk-0.12.1.dist-info/top_level.txt,sha256=o41t5TzwgoxzSmbKoP6olWW1FyEAGWVYTjeoKadBK40,13
|
|
22
|
+
periplus_python_sdk-0.12.1.dist-info/RECORD,,
|
|
@@ -7,7 +7,7 @@ The Periplus license does not replace these terms.
|
|
|
7
7
|
|
|
8
8
|
## Selectolax and Lexbor
|
|
9
9
|
|
|
10
|
-
The backend pins Selectolax 0.4.
|
|
10
|
+
The backend pins Selectolax 0.4.12 (MIT) and its bundled Lexbor 3.1.0
|
|
11
11
|
(Apache-2.0). The DOM adapter's read-only native layouts follow Lexbor's DOM and
|
|
12
12
|
HTML interface headers. Selectolax retains its packaged MIT license; Periplus
|
|
13
13
|
includes Lexbor's license and notice under `licensing/third-party/lexbor/`
|
periplus_sdk/__init__.py
CHANGED
|
@@ -1,9 +1,13 @@
|
|
|
1
|
-
"""
|
|
1
|
+
"""Typed organization operations and read-only SQL through the Periplus API."""
|
|
2
2
|
from .dbapi import connect
|
|
3
3
|
from .client import AsyncClient, Client
|
|
4
4
|
from .errors import ApiError, ConfigurationError, PeriplusError, ResponseError, TransportError
|
|
5
|
-
from .types import Diagnostic, PreparedQuery, QueryHelper, QueryHelpers, QueryResult
|
|
5
|
+
from .types import Diagnostic, PreparedQuery, QueryHelper, QueryHelpers, QueryResult, SourceSnapshot
|
|
6
|
+
from .resources_types import (FollowRule, QueryCondition, Page, Request, RequestPage, Pin, PinCapture, Schedule, SchedulePage, Member,
|
|
7
|
+
Invitation, ApiKey, CreatedApiKey, Identity, Usage)
|
|
6
8
|
|
|
7
9
|
__all__ = ["connect", "AsyncClient", "Client", "ApiError", "ConfigurationError", "PeriplusError",
|
|
8
10
|
"ResponseError", "TransportError", "Diagnostic", "PreparedQuery", "QueryHelper",
|
|
9
|
-
"QueryHelpers", "QueryResult"
|
|
11
|
+
"QueryHelpers", "QueryResult", "FollowRule", "QueryCondition", "Page", "Request", "RequestPage", "Pin", "PinCapture",
|
|
12
|
+
"Schedule", "SchedulePage", "Member", "Invitation",
|
|
13
|
+
"ApiKey", "CreatedApiKey", "Identity", "Usage", "SourceSnapshot"]
|
|
@@ -0,0 +1,366 @@
|
|
|
1
|
+
"""Async customer operations; the same contract as resources.py."""
|
|
2
|
+
from __future__ import annotations
|
|
3
|
+
from collections.abc import AsyncIterator, Sequence
|
|
4
|
+
from datetime import date, datetime
|
|
5
|
+
from typing import Generic, Literal, TypeVar
|
|
6
|
+
from urllib.parse import quote
|
|
7
|
+
from uuid import UUID, uuid4
|
|
8
|
+
|
|
9
|
+
from pydantic import JsonValue
|
|
10
|
+
|
|
11
|
+
from .resources_types import (
|
|
12
|
+
ApiKey, AuditEvent, Availability, CreatedApiKey, Identity, Invitation, Member, Members,
|
|
13
|
+
FollowRule, Page, Pin, PinCapture, QueryExecution, QueryUsage, Request, RequestPagesPage, Role,
|
|
14
|
+
SavedQuery, Schedule, SchedulePage, Usage,
|
|
15
|
+
)
|
|
16
|
+
|
|
17
|
+
|
|
18
|
+
def path_id(value: str | UUID) -> str:
|
|
19
|
+
return quote(str(value), safe="")
|
|
20
|
+
|
|
21
|
+
|
|
22
|
+
def page_offset(cursor: str | None) -> int:
|
|
23
|
+
if cursor is None:
|
|
24
|
+
return 0
|
|
25
|
+
if not cursor.isascii() or not cursor.isdecimal():
|
|
26
|
+
raise ValueError("Invalid page cursor")
|
|
27
|
+
return int(cursor)
|
|
28
|
+
|
|
29
|
+
|
|
30
|
+
def page_size(limit: int) -> int:
|
|
31
|
+
if not 1 <= limit <= 100:
|
|
32
|
+
raise ValueError("Page size must be between 1 and 100")
|
|
33
|
+
return limit
|
|
34
|
+
|
|
35
|
+
|
|
36
|
+
def versioned_ref(value: Pin | Schedule | str | UUID, expected_version: int | None, model: type):
|
|
37
|
+
if isinstance(value, (Pin, Schedule)):
|
|
38
|
+
if not isinstance(value, model):
|
|
39
|
+
raise ValueError(f"Expected a {model.__name__}")
|
|
40
|
+
return value.id, value.version if expected_version is None else expected_version
|
|
41
|
+
if expected_version is None:
|
|
42
|
+
raise ValueError("An ID requires expected_version; fetch the resource before changing it")
|
|
43
|
+
return value, expected_version
|
|
44
|
+
|
|
45
|
+
|
|
46
|
+
def object_id(value) -> str:
|
|
47
|
+
return path_id(value.id if isinstance(value, (Request, Pin, Schedule)) else value)
|
|
48
|
+
|
|
49
|
+
|
|
50
|
+
def selection(ids_name: str, ids: Sequence | None, sql: str | None, parameters: Sequence[JsonValue]) -> dict:
|
|
51
|
+
if (ids is None) == (sql is None):
|
|
52
|
+
raise ValueError(f"Choose either {ids_name} or sql")
|
|
53
|
+
if sql is None:
|
|
54
|
+
if parameters:
|
|
55
|
+
raise ValueError("parameters require sql")
|
|
56
|
+
return {ids_name: [str(value) for value in ids]}
|
|
57
|
+
return {"sql": sql, "parameters": list(parameters)}
|
|
58
|
+
|
|
59
|
+
|
|
60
|
+
def member_ref(member: Member | str, expected_role: Role | None):
|
|
61
|
+
if isinstance(member, Member):
|
|
62
|
+
return member.id, expected_role if expected_role is not None else member.role
|
|
63
|
+
if expected_role is None:
|
|
64
|
+
raise ValueError("An ID requires expected_role; fetch membership before changing it")
|
|
65
|
+
return member, expected_role
|
|
66
|
+
|
|
67
|
+
|
|
68
|
+
class Resource:
|
|
69
|
+
def __init__(self, client):
|
|
70
|
+
self._client = client
|
|
71
|
+
|
|
72
|
+
T = TypeVar("T")
|
|
73
|
+
|
|
74
|
+
|
|
75
|
+
class PagedResource(Resource, Generic[T]):
|
|
76
|
+
async def iter(self, **filters) -> AsyncIterator[T]:
|
|
77
|
+
"""Lazily iterate list pages. No snapshot is implied while resources change."""
|
|
78
|
+
cursor = filters.pop("cursor", None)
|
|
79
|
+
while True:
|
|
80
|
+
page = await self.list(cursor=cursor, **filters)
|
|
81
|
+
for item in page.items:
|
|
82
|
+
yield item
|
|
83
|
+
if page.next_cursor is None:
|
|
84
|
+
return
|
|
85
|
+
if page.next_cursor == cursor:
|
|
86
|
+
raise ValueError("Server returned a non-advancing cursor")
|
|
87
|
+
cursor = page.next_cursor
|
|
88
|
+
|
|
89
|
+
|
|
90
|
+
class IdentityResource(Resource):
|
|
91
|
+
async def get(self) -> Identity:
|
|
92
|
+
return await self._client._resource_request("GET", "identity", Identity)
|
|
93
|
+
|
|
94
|
+
|
|
95
|
+
class AvailabilityResource(Resource):
|
|
96
|
+
async def get(self) -> Availability:
|
|
97
|
+
return await self._client._resource_request("GET", "access", Availability)
|
|
98
|
+
|
|
99
|
+
|
|
100
|
+
class RequestResource(PagedResource[Request]):
|
|
101
|
+
"""Shared reads for captures and crawls: one request view and one page feed."""
|
|
102
|
+
noun: Literal["captures", "crawls"]
|
|
103
|
+
|
|
104
|
+
async def list(self, *, finished: bool | None = None, cursor: str | None = None,
|
|
105
|
+
limit: int = 20) -> Page[Request]:
|
|
106
|
+
return await self._client._resource_request("GET", self.noun, Page[Request],
|
|
107
|
+
params={"finished": finished, "offset": page_offset(cursor), "limit": page_size(limit)})
|
|
108
|
+
|
|
109
|
+
async def get(self, id: Request | str | UUID, *, wait_seconds: int = 0) -> Request:
|
|
110
|
+
"""Return as soon as the request is finished and queryable, or after `wait_seconds` (0-30).
|
|
111
|
+
|
|
112
|
+
Waiting never delays capture. The SDK does not poll; call again to keep waiting."""
|
|
113
|
+
if not 0 <= wait_seconds <= 30:
|
|
114
|
+
raise ValueError("wait_seconds must be between 0 and 30")
|
|
115
|
+
return await self._client._resource_request("GET", f"{self.noun}/{object_id(id)}", Request,
|
|
116
|
+
params={"wait_seconds": wait_seconds or None})
|
|
117
|
+
|
|
118
|
+
async def pages(self, id: Request | str | UUID, *,
|
|
119
|
+
status: Literal["captured", "failed", "pending"] | None = None,
|
|
120
|
+
search: str | None = None, host: str | None = None, depth: int | None = None,
|
|
121
|
+
failure_code: str | None = None, render_incomplete: bool | None = None,
|
|
122
|
+
cursor: str | None = None, limit: int = 20) -> RequestPagesPage:
|
|
123
|
+
"""Requested pages, newest first; `capture_id` is the SQL key. Pass cursors unchanged.
|
|
124
|
+
|
|
125
|
+
Failure-code filtering scans a bounded window, so an empty page can still have a cursor.
|
|
126
|
+
`render_incomplete=True` lists captured pages whose HTML was only a loading placeholder."""
|
|
127
|
+
return await self._client._resource_request("GET", f"{self.noun}/{object_id(id)}/pages", RequestPagesPage,
|
|
128
|
+
params={"status": status, "search": search, "host": host, "depth": depth,
|
|
129
|
+
"failure_code": failure_code, "render_incomplete": render_incomplete,
|
|
130
|
+
"cursor": cursor, "limit": page_size(limit)})
|
|
131
|
+
|
|
132
|
+
async def cancel(self, id: Request | str | UUID) -> Request:
|
|
133
|
+
"""Stop remaining work. Captures already made are kept."""
|
|
134
|
+
return await self._client._resource_request("POST", f"{self.noun}/{object_id(id)}/cancel", Request)
|
|
135
|
+
|
|
136
|
+
|
|
137
|
+
class CapturesResource(RequestResource):
|
|
138
|
+
noun = "captures"
|
|
139
|
+
|
|
140
|
+
async def create(self, *, urls: Sequence[str], pin_days: int | None = None,
|
|
141
|
+
id: UUID | None = None) -> Request:
|
|
142
|
+
"""Fetch 1-100 known URLs once, now.
|
|
143
|
+
|
|
144
|
+
Always fetches again and never returns an existing capture: check coverage with SQL
|
|
145
|
+
first. `pin_days` pins each successful capture (requires organization:pins:write).
|
|
146
|
+
Keep `id` to reconcile an uncertain write; the same ID and input return the same request."""
|
|
147
|
+
return await self._client._resource_request("POST", "captures", Request, json={
|
|
148
|
+
"id": str(id or uuid4()), "urls": list(urls),
|
|
149
|
+
**({"pin_days": pin_days} if pin_days is not None else {})})
|
|
150
|
+
|
|
151
|
+
|
|
152
|
+
class CrawlsResource(RequestResource):
|
|
153
|
+
noun = "crawls"
|
|
154
|
+
|
|
155
|
+
async def admission_preview(self, *, seeds: Sequence[str], accepted_queue_wait_seconds: int | None = None) -> Availability:
|
|
156
|
+
return await self._client._resource_request("POST", "access/crawl-preview", Availability, json={
|
|
157
|
+
"seed_urls": list(seeds), "accepted_queue_wait_seconds": accepted_queue_wait_seconds})
|
|
158
|
+
|
|
159
|
+
async def create(self, *, seeds: Sequence[str], page_limit: int, max_depth: int = 2,
|
|
160
|
+
follow_scope: Literal["starting_sites", "linked_sites"] = "starting_sites",
|
|
161
|
+
allowed_hosts: Sequence[str] = (), excluded_hosts: Sequence[str] = (),
|
|
162
|
+
allowed_paths: Sequence[str] = (), excluded_paths: Sequence[str] = (),
|
|
163
|
+
follow_rules: Sequence[FollowRule | dict] = (),
|
|
164
|
+
accepted_queue_wait_seconds: int | None = None,
|
|
165
|
+
pin_days: int | None = None, id: UUID | None = None) -> Request:
|
|
166
|
+
"""Follow links from seeds when the URLs are not known yet.
|
|
167
|
+
|
|
168
|
+
`page_limit` means up to N pages including seeds, not guaranteed coverage. `max_depth`
|
|
169
|
+
is at least 1; use `captures.create` for known URLs. Hosts include subdomains; path
|
|
170
|
+
patterns match the whole URL-encoded path and only `*` is special."""
|
|
171
|
+
return await self._client._resource_request("POST", "crawls", Request, json={
|
|
172
|
+
"id": str(id or uuid4()), "seeds": list(seeds), "page_limit": page_limit,
|
|
173
|
+
"max_depth": max_depth, "follow_scope": follow_scope,
|
|
174
|
+
**({"accepted_queue_wait_seconds": accepted_queue_wait_seconds} if accepted_queue_wait_seconds is not None else {}),
|
|
175
|
+
"allowed_hosts": list(allowed_hosts), "excluded_hosts": list(excluded_hosts),
|
|
176
|
+
"allowed_paths": list(allowed_paths), "excluded_paths": list(excluded_paths),
|
|
177
|
+
"follow_rules": [rule.model_dump(exclude_none=True) if isinstance(rule, FollowRule) else rule
|
|
178
|
+
for rule in follow_rules],
|
|
179
|
+
**({"pin_days": pin_days} if pin_days is not None else {})})
|
|
180
|
+
|
|
181
|
+
|
|
182
|
+
class PinsResource(PagedResource[Pin]):
|
|
183
|
+
async def create(self, *, name: str, days: int, capture_ids: Sequence[str | UUID] | None = None,
|
|
184
|
+
sql: str | None = None, parameters: Sequence[JsonValue] = (),
|
|
185
|
+
saved_query_id: str | UUID | None = None, id: UUID | None = None) -> Pin:
|
|
186
|
+
"""Keep exact captures for `days`, active immediately.
|
|
187
|
+
|
|
188
|
+
Pass `capture_ids`, one read-only `sql` query returning `capture_id`, or the
|
|
189
|
+
`saved_query_id` of such a query in your organization. SQL runs once at creation and its
|
|
190
|
+
exact result is locked in; running it beforehand is only an estimate. Membership never
|
|
191
|
+
changes. SQL selection also requires organization:sql:exec; a saved query also
|
|
192
|
+
requires organization:sql:read."""
|
|
193
|
+
if saved_query_id is not None:
|
|
194
|
+
if capture_ids is not None or sql is not None:
|
|
195
|
+
raise ValueError("Choose one of capture_ids, sql or saved_query_id")
|
|
196
|
+
source = {"saved_query_id": str(saved_query_id), "parameters": list(parameters)}
|
|
197
|
+
else:
|
|
198
|
+
source = selection("capture_ids", capture_ids, sql, parameters)
|
|
199
|
+
return await self._client._resource_request("POST", "pins", Pin, json={
|
|
200
|
+
"id": str(id or uuid4()), "name": name, "days": days, **source})
|
|
201
|
+
|
|
202
|
+
async def list(self, *, cursor: str | None = None, limit: int = 20) -> Page[Pin]:
|
|
203
|
+
return await self._client._resource_request("GET", "pins", Page[Pin],
|
|
204
|
+
params={"offset": page_offset(cursor), "limit": page_size(limit)})
|
|
205
|
+
|
|
206
|
+
async def get(self, id: Pin | str | UUID) -> Pin:
|
|
207
|
+
return await self._client._resource_request("GET", f"pins/{object_id(id)}", Pin)
|
|
208
|
+
|
|
209
|
+
async def captures(self, id: Pin | str | UUID, *, cursor: str | None = None,
|
|
210
|
+
limit: int = 100) -> Page[PinCapture]:
|
|
211
|
+
"""Pinned captures with each one's expiry."""
|
|
212
|
+
return await self._client._resource_request("GET", f"pins/{object_id(id)}/captures", Page[PinCapture],
|
|
213
|
+
params={"offset": page_offset(cursor), "limit": page_size(limit)})
|
|
214
|
+
|
|
215
|
+
async def update(self, pin: Pin | str | UUID, *, name: str | None = None, days: int | None = None,
|
|
216
|
+
expected_version: int | None = None) -> Pin:
|
|
217
|
+
"""Rename, or change `days`: every capture's expiry moves by the difference."""
|
|
218
|
+
id, version = versioned_ref(pin, expected_version, Pin)
|
|
219
|
+
return await self._client._resource_request("PATCH", f"pins/{path_id(id)}", Pin, json={
|
|
220
|
+
"expected_version": version,
|
|
221
|
+
**{key: value for key, value in {"name": name, "days": days}.items() if value is not None}})
|
|
222
|
+
|
|
223
|
+
async def delete(self, id: Pin | str | UUID) -> None:
|
|
224
|
+
"""Remove future protection. Captures and past usage are not deleted."""
|
|
225
|
+
return await self._client._resource_request("DELETE", f"pins/{object_id(id)}", None)
|
|
226
|
+
|
|
227
|
+
|
|
228
|
+
class SchedulesResource(PagedResource[Schedule]):
|
|
229
|
+
async def create(self, *, name: str, every_days: int, urls: Sequence[str] | None = None,
|
|
230
|
+
sql: str | None = None, parameters: Sequence[JsonValue] = (),
|
|
231
|
+
pin_days: int | None = None, id: UUID | None = None) -> Schedule:
|
|
232
|
+
"""Capture a fixed URL list every `every_days`, active immediately.
|
|
233
|
+
|
|
234
|
+
Pass `urls` (they need not be in the corpus) or one read-only `sql` query returning a
|
|
235
|
+
`url` column; SQL runs once and its URLs are locked in. Scheduled runs follow no links.
|
|
236
|
+
`pin_days` pins every successful scheduled capture (requires organization:pins:write)."""
|
|
237
|
+
return await self._client._resource_request("POST", "schedules", Schedule, json={
|
|
238
|
+
"id": str(id or uuid4()), "name": name, "every_days": every_days,
|
|
239
|
+
**selection("urls", urls, sql, parameters),
|
|
240
|
+
**({"pin_days": pin_days} if pin_days is not None else {})})
|
|
241
|
+
|
|
242
|
+
async def list(self, *, cursor: str | None = None, limit: int = 20) -> Page[Schedule]:
|
|
243
|
+
return await self._client._resource_request("GET", "schedules", Page[Schedule],
|
|
244
|
+
params={"offset": page_offset(cursor), "limit": page_size(limit)})
|
|
245
|
+
|
|
246
|
+
async def get(self, id: Schedule | str | UUID) -> Schedule:
|
|
247
|
+
return await self._client._resource_request("GET", f"schedules/{object_id(id)}", Schedule)
|
|
248
|
+
|
|
249
|
+
async def pages(self, id: Schedule | str | UUID, *, search: str | None = None,
|
|
250
|
+
cursor: str | None = None, limit: int = 100) -> Page[SchedulePage]:
|
|
251
|
+
"""Scheduled URLs with their latest success, failure and next due time."""
|
|
252
|
+
return await self._client._resource_request("GET", f"schedules/{object_id(id)}/pages", Page[SchedulePage],
|
|
253
|
+
params={"search": search, "offset": page_offset(cursor), "limit": page_size(limit)})
|
|
254
|
+
|
|
255
|
+
async def captures(self, id: Schedule | str | UUID, *, status: Literal["captured", "failed"] | None = None,
|
|
256
|
+
search: str | None = None, render_incomplete: bool | None = None,
|
|
257
|
+
cursor: str | None = None, limit: int = 20) -> RequestPagesPage:
|
|
258
|
+
"""Results of scheduled runs, newest first, in the request page-feed shape."""
|
|
259
|
+
return await self._client._resource_request("GET", f"schedules/{object_id(id)}/captures", RequestPagesPage,
|
|
260
|
+
params={"status": status, "search": search, "render_incomplete": render_incomplete,
|
|
261
|
+
"cursor": cursor, "limit": page_size(limit)})
|
|
262
|
+
|
|
263
|
+
async def update(self, schedule: Schedule | str | UUID, *, name: str | None = None,
|
|
264
|
+
every_days: int | None = None, expected_version: int | None = None) -> Schedule:
|
|
265
|
+
id, version = versioned_ref(schedule, expected_version, Schedule)
|
|
266
|
+
return await self._client._resource_request("PATCH", f"schedules/{path_id(id)}", Schedule, json={
|
|
267
|
+
"expected_version": version, **{key: value for key, value in
|
|
268
|
+
{"name": name, "every_days": every_days}.items() if value is not None}})
|
|
269
|
+
|
|
270
|
+
async def pause(self, id: Schedule | str | UUID) -> Schedule:
|
|
271
|
+
"""Stop new runs; an unstarted run is cancelled, started captures finish."""
|
|
272
|
+
return await self._client._resource_request("POST", f"schedules/{object_id(id)}/pause", Schedule)
|
|
273
|
+
|
|
274
|
+
async def resume(self, id: Schedule | str | UUID) -> Schedule:
|
|
275
|
+
return await self._client._resource_request("POST", f"schedules/{object_id(id)}/resume", Schedule)
|
|
276
|
+
|
|
277
|
+
async def delete(self, id: Schedule | str | UUID) -> None:
|
|
278
|
+
"""Remove the schedule. Captures it produced are kept."""
|
|
279
|
+
return await self._client._resource_request("DELETE", f"schedules/{object_id(id)}", None)
|
|
280
|
+
|
|
281
|
+
|
|
282
|
+
class SavedQueriesResource(PagedResource[SavedQuery]):
|
|
283
|
+
async def list(self, *, search: str = "", cursor: str | None = None) -> Page[SavedQuery]:
|
|
284
|
+
return await self._client._resource_request("GET", "saved-queries", Page[SavedQuery],
|
|
285
|
+
params={"search": search, "offset": page_offset(cursor)})
|
|
286
|
+
|
|
287
|
+
async def get(self, id: str | UUID) -> SavedQuery:
|
|
288
|
+
return await self._client._resource_request("GET", f"saved-queries/{path_id(id)}", SavedQuery)
|
|
289
|
+
|
|
290
|
+
async def create(self, *, name: str, sql: str, id: UUID | None = None) -> SavedQuery:
|
|
291
|
+
return await self._client._resource_request("POST", "saved-queries", SavedQuery,
|
|
292
|
+
json={"id": str(id or uuid4()), "name": name, "sql": sql})
|
|
293
|
+
|
|
294
|
+
async def rename(self, id: str | UUID, *, name: str) -> SavedQuery:
|
|
295
|
+
return await self._client._resource_request("PATCH", f"saved-queries/{path_id(id)}", SavedQuery, json={"name": name})
|
|
296
|
+
|
|
297
|
+
async def delete(self, id: str | UUID) -> None:
|
|
298
|
+
return await self._client._resource_request("DELETE", f"saved-queries/{path_id(id)}", None)
|
|
299
|
+
|
|
300
|
+
|
|
301
|
+
class MembersResource(Resource):
|
|
302
|
+
async def list(self) -> Members:
|
|
303
|
+
"""Organization membership and invitation snapshot, including owner guards."""
|
|
304
|
+
return await self._client._resource_request("GET", "members", Members)
|
|
305
|
+
|
|
306
|
+
async def update_role(self, member: Member | str, *, role: Role, expected_role: Role | None = None) -> None:
|
|
307
|
+
member_id, current_role = member_ref(member, expected_role)
|
|
308
|
+
return await self._client._resource_request("PUT", f"members/{path_id(member_id)}", None,
|
|
309
|
+
json={"action": role, "expected_role": current_role})
|
|
310
|
+
|
|
311
|
+
async def remove(self, member: Member | str, *, expected_role: Role | None = None) -> None:
|
|
312
|
+
member_id, current_role = member_ref(member, expected_role)
|
|
313
|
+
return await self._client._resource_request("PUT", f"members/{path_id(member_id)}", None,
|
|
314
|
+
json={"action": "remove", "expected_role": current_role})
|
|
315
|
+
|
|
316
|
+
|
|
317
|
+
class InvitationsResource(Resource):
|
|
318
|
+
async def list(self) -> list[Invitation]:
|
|
319
|
+
return (await self._client._resource_request("GET", "members", Members)).invitations
|
|
320
|
+
|
|
321
|
+
async def create(self, *, email: str, role: Role = "developer") -> None:
|
|
322
|
+
return await self._client._resource_request("POST", "members/invitations", None,
|
|
323
|
+
json={"email": email, "role": role})
|
|
324
|
+
|
|
325
|
+
async def cancel(self, id: str) -> None:
|
|
326
|
+
return await self._client._resource_request("DELETE", f"members/invitations/{path_id(id)}", None)
|
|
327
|
+
|
|
328
|
+
|
|
329
|
+
class ApiKeysResource(PagedResource[ApiKey]):
|
|
330
|
+
async def list(self, *, cursor: str | None = None, limit: int = 20) -> Page[ApiKey]:
|
|
331
|
+
return await self._client._resource_request("GET", "api-keys", Page[ApiKey],
|
|
332
|
+
params={"offset": page_offset(cursor), "limit": page_size(limit)})
|
|
333
|
+
|
|
334
|
+
async def create(self, *, name: str, id: UUID | None = None, profile: Literal["developer", "agent"] = "developer", expires_at: datetime | None = None) -> CreatedApiKey:
|
|
335
|
+
"""Secret is returned once; access via secret.get_secret_value()."""
|
|
336
|
+
return await self._client._resource_request("POST", "api-keys", CreatedApiKey,
|
|
337
|
+
json={"id": str(id or uuid4()), "name": name, **({"profile": profile} if profile != "developer" else {}),
|
|
338
|
+
**({"expires_at": expires_at.isoformat()} if expires_at is not None else {})})
|
|
339
|
+
|
|
340
|
+
async def revoke(self, id: str | UUID) -> None:
|
|
341
|
+
return await self._client._resource_request("DELETE", f"api-keys/{path_id(id)}", None)
|
|
342
|
+
|
|
343
|
+
|
|
344
|
+
class UsageResource(Resource):
|
|
345
|
+
async def get(self, *, start: date | None = None, end: date | None = None) -> Usage:
|
|
346
|
+
"""UTC dates; end is exclusive. Usage is consumption, not an invoice."""
|
|
347
|
+
return await self._client._resource_request("GET", "usage", Usage, params={"start": start, "end": end})
|
|
348
|
+
|
|
349
|
+
|
|
350
|
+
class QueryHistoryResource(PagedResource[QueryExecution]):
|
|
351
|
+
async def summary(self, *, start: date | None = None, end: date | None = None) -> QueryUsage:
|
|
352
|
+
return await self._client._resource_request("GET", "usage/queries", QueryUsage, params={"start": start, "end": end})
|
|
353
|
+
|
|
354
|
+
async def list(self, *, cursor: str | None = None, limit: int = 50) -> Page[QueryExecution]:
|
|
355
|
+
return await self._client._resource_request("GET", "usage/query-history", Page[QueryExecution],
|
|
356
|
+
params={"cursor": cursor, "limit": page_size(limit)})
|
|
357
|
+
|
|
358
|
+
async def get(self, id: str | UUID) -> QueryExecution:
|
|
359
|
+
return await self._client._resource_request("GET", f"usage/queries/{path_id(id)}", QueryExecution)
|
|
360
|
+
|
|
361
|
+
|
|
362
|
+
class AuditResource(PagedResource[AuditEvent]):
|
|
363
|
+
async def list(self, *, cursor: str | None = None, action: str | None = None,
|
|
364
|
+
outcome: str | None = None) -> Page[AuditEvent]:
|
|
365
|
+
return await self._client._resource_request("GET", "audit", Page[AuditEvent],
|
|
366
|
+
params={"offset": page_offset(cursor), "action": action, "outcome": outcome})
|