apify-client 3.2.0__tar.gz → 3.2.1b2__tar.gz
This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
- {apify_client-3.2.0 → apify_client-3.2.1b2}/CHANGELOG.md +9 -0
- {apify_client-3.2.0 → apify_client-3.2.1b2}/PKG-INFO +3 -3
- {apify_client-3.2.0 → apify_client-3.2.1b2}/README.md +2 -2
- {apify_client-3.2.0 → apify_client-3.2.1b2}/pyproject.toml +1 -1
- {apify_client-3.2.0 → apify_client-3.2.1b2}/pyproject.toml.orig +1 -1
- {apify_client-3.2.0 → apify_client-3.2.1b2}/src/apify_client/_pagination.py +54 -26
- {apify_client-3.2.0 → apify_client-3.2.1b2}/src/apify_client/_resource_clients/dataset.py +11 -9
- {apify_client-3.2.0 → apify_client-3.2.1b2}/src/apify_client/_streamed_log.py +3 -3
- {apify_client-3.2.0 → apify_client-3.2.1b2}/src/apify_client/http_clients/__init__.py +1 -1
- {apify_client-3.2.0 → apify_client-3.2.1b2}/src/apify_client/http_clients/_httpx2.py +47 -47
- {apify_client-3.2.0 → apify_client-3.2.1b2}/CONTRIBUTING.md +0 -0
- {apify_client-3.2.0 → apify_client-3.2.1b2}/LICENSE +0 -0
- {apify_client-3.2.0 → apify_client-3.2.1b2}/src/apify_client/__init__.py +0 -0
- {apify_client-3.2.0 → apify_client-3.2.1b2}/src/apify_client/_apify_client.py +0 -0
- {apify_client-3.2.0 → apify_client-3.2.1b2}/src/apify_client/_client_registry.py +0 -0
- {apify_client-3.2.0 → apify_client-3.2.1b2}/src/apify_client/_consts.py +0 -0
- {apify_client-3.2.0 → apify_client-3.2.1b2}/src/apify_client/_docs.py +0 -0
- {apify_client-3.2.0 → apify_client-3.2.1b2}/src/apify_client/_literals.py +0 -0
- {apify_client-3.2.0 → apify_client-3.2.1b2}/src/apify_client/_logging.py +0 -0
- {apify_client-3.2.0 → apify_client-3.2.1b2}/src/apify_client/_models.py +0 -0
- {apify_client-3.2.0 → apify_client-3.2.1b2}/src/apify_client/_resource_clients/__init__.py +0 -0
- {apify_client-3.2.0 → apify_client-3.2.1b2}/src/apify_client/_resource_clients/_resource_client.py +0 -0
- {apify_client-3.2.0 → apify_client-3.2.1b2}/src/apify_client/_resource_clients/actor.py +0 -0
- {apify_client-3.2.0 → apify_client-3.2.1b2}/src/apify_client/_resource_clients/actor_collection.py +0 -0
- {apify_client-3.2.0 → apify_client-3.2.1b2}/src/apify_client/_resource_clients/actor_env_var.py +0 -0
- {apify_client-3.2.0 → apify_client-3.2.1b2}/src/apify_client/_resource_clients/actor_env_var_collection.py +0 -0
- {apify_client-3.2.0 → apify_client-3.2.1b2}/src/apify_client/_resource_clients/actor_version.py +0 -0
- {apify_client-3.2.0 → apify_client-3.2.1b2}/src/apify_client/_resource_clients/actor_version_collection.py +0 -0
- {apify_client-3.2.0 → apify_client-3.2.1b2}/src/apify_client/_resource_clients/build.py +0 -0
- {apify_client-3.2.0 → apify_client-3.2.1b2}/src/apify_client/_resource_clients/build_collection.py +0 -0
- {apify_client-3.2.0 → apify_client-3.2.1b2}/src/apify_client/_resource_clients/dataset_collection.py +0 -0
- {apify_client-3.2.0 → apify_client-3.2.1b2}/src/apify_client/_resource_clients/key_value_store.py +0 -0
- {apify_client-3.2.0 → apify_client-3.2.1b2}/src/apify_client/_resource_clients/key_value_store_collection.py +0 -0
- {apify_client-3.2.0 → apify_client-3.2.1b2}/src/apify_client/_resource_clients/log.py +0 -0
- {apify_client-3.2.0 → apify_client-3.2.1b2}/src/apify_client/_resource_clients/request_queue.py +0 -0
- {apify_client-3.2.0 → apify_client-3.2.1b2}/src/apify_client/_resource_clients/request_queue_collection.py +0 -0
- {apify_client-3.2.0 → apify_client-3.2.1b2}/src/apify_client/_resource_clients/run.py +0 -0
- {apify_client-3.2.0 → apify_client-3.2.1b2}/src/apify_client/_resource_clients/run_collection.py +0 -0
- {apify_client-3.2.0 → apify_client-3.2.1b2}/src/apify_client/_resource_clients/schedule.py +0 -0
- {apify_client-3.2.0 → apify_client-3.2.1b2}/src/apify_client/_resource_clients/schedule_collection.py +0 -0
- {apify_client-3.2.0 → apify_client-3.2.1b2}/src/apify_client/_resource_clients/store_collection.py +0 -0
- {apify_client-3.2.0 → apify_client-3.2.1b2}/src/apify_client/_resource_clients/task.py +0 -0
- {apify_client-3.2.0 → apify_client-3.2.1b2}/src/apify_client/_resource_clients/task_collection.py +0 -0
- {apify_client-3.2.0 → apify_client-3.2.1b2}/src/apify_client/_resource_clients/user.py +0 -0
- {apify_client-3.2.0 → apify_client-3.2.1b2}/src/apify_client/_resource_clients/webhook.py +0 -0
- {apify_client-3.2.0 → apify_client-3.2.1b2}/src/apify_client/_resource_clients/webhook_collection.py +0 -0
- {apify_client-3.2.0 → apify_client-3.2.1b2}/src/apify_client/_resource_clients/webhook_dispatch.py +0 -0
- {apify_client-3.2.0 → apify_client-3.2.1b2}/src/apify_client/_resource_clients/webhook_dispatch_collection.py +0 -0
- {apify_client-3.2.0 → apify_client-3.2.1b2}/src/apify_client/_statistics.py +0 -0
- {apify_client-3.2.0 → apify_client-3.2.1b2}/src/apify_client/_status_message_watcher.py +0 -0
- {apify_client-3.2.0 → apify_client-3.2.1b2}/src/apify_client/_typeddicts.py +0 -0
- {apify_client-3.2.0 → apify_client-3.2.1b2}/src/apify_client/_utils/__init__.py +0 -0
- {apify_client-3.2.0 → apify_client-3.2.1b2}/src/apify_client/_utils/crypto.py +0 -0
- {apify_client-3.2.0 → apify_client-3.2.1b2}/src/apify_client/_utils/encoding.py +0 -0
- {apify_client-3.2.0 → apify_client-3.2.1b2}/src/apify_client/_utils/errors.py +0 -0
- {apify_client-3.2.0 → apify_client-3.2.1b2}/src/apify_client/_utils/http.py +0 -0
- {apify_client-3.2.0 → apify_client-3.2.1b2}/src/apify_client/_utils/time.py +0 -0
- {apify_client-3.2.0 → apify_client-3.2.1b2}/src/apify_client/_utils/try_import.py +0 -0
- {apify_client-3.2.0 → apify_client-3.2.1b2}/src/apify_client/errors.py +0 -0
- {apify_client-3.2.0 → apify_client-3.2.1b2}/src/apify_client/http_clients/_base.py +0 -0
- {apify_client-3.2.0 → apify_client-3.2.1b2}/src/apify_client/http_clients/_impit.py +0 -0
- {apify_client-3.2.0 → apify_client-3.2.1b2}/src/apify_client/http_compressors/__init__.py +0 -0
- {apify_client-3.2.0 → apify_client-3.2.1b2}/src/apify_client/http_compressors/_base.py +0 -0
- {apify_client-3.2.0 → apify_client-3.2.1b2}/src/apify_client/http_compressors/_brotli.py +0 -0
- {apify_client-3.2.0 → apify_client-3.2.1b2}/src/apify_client/http_compressors/_gzip.py +0 -0
- {apify_client-3.2.0 → apify_client-3.2.1b2}/src/apify_client/http_compressors/_resolve.py +0 -0
- {apify_client-3.2.0 → apify_client-3.2.1b2}/src/apify_client/py.typed +0 -0
- {apify_client-3.2.0 → apify_client-3.2.1b2}/src/apify_client/types.py +0 -0
|
@@ -2,6 +2,15 @@
|
|
|
2
2
|
|
|
3
3
|
All notable changes to this project will be documented in this file.
|
|
4
4
|
|
|
5
|
+
<!-- git-cliff-unreleased-start -->
|
|
6
|
+
## 3.2.1 - **not yet released**
|
|
7
|
+
|
|
8
|
+
### 🐛 Bug Fixes
|
|
9
|
+
|
|
10
|
+
- Stop dataset iterators from skipping items when unwind is used ([#1059](https://github.com/apify/apify-client-python/pull/1059)) ([8e30197](https://github.com/apify/apify-client-python/commit/8e301977ecb80b7c01251cacf071a3f6201469ba)) by [@vdusek](https://github.com/vdusek), closes [#1058](https://github.com/apify/apify-client-python/issues/1058)
|
|
11
|
+
|
|
12
|
+
|
|
13
|
+
<!-- git-cliff-unreleased-end -->
|
|
5
14
|
## [3.2.0](https://github.com/apify/apify-client-python/releases/tag/v3.2.0) (2026-09-03)
|
|
6
15
|
|
|
7
16
|
### 🚀 Features
|
|
@@ -1,6 +1,6 @@
|
|
|
1
1
|
Metadata-Version: 2.4
|
|
2
2
|
Name: apify_client
|
|
3
|
-
Version: 3.2.
|
|
3
|
+
Version: 3.2.1b2
|
|
4
4
|
Summary: Apify API client for Python
|
|
5
5
|
Keywords: apify,api,client,automation,crawling,scraping
|
|
6
6
|
Author: Apify Technologies s.r.o.
|
|
@@ -95,7 +95,7 @@ Description-Content-Type: text/markdown
|
|
|
95
95
|
```
|
|
96
96
|
|
|
97
97
|
[Impit](https://github.com/apify/impit) is the default HTTP client and is installed automatically. To use the
|
|
98
|
-
built-in [
|
|
98
|
+
built-in [HTTPX2](https://github.com/pydantic/httpx2) client instead, install the optional `httpx2` extra, which
|
|
99
99
|
provides Pydantic's maintained continuation of HTTPX, and pass `http_client=Httpx2HttpClient()` to
|
|
100
100
|
`ApifyClient.with_custom_http_client()`:
|
|
101
101
|
|
|
@@ -172,7 +172,7 @@ For a guided walkthrough — authenticating, running an Actor, and reading its r
|
|
|
172
172
|
- **Tiered timeouts** — short / medium / long tiers picked per endpoint, overridable per call ([Timeouts](https://docs.apify.com/api/client/python/docs/concepts/timeouts)).
|
|
173
173
|
- **Pagination and streaming** — iterate datasets, key-value store keys, or live logs without manual paging or buffering ([Pagination](https://docs.apify.com/api/client/python/docs/concepts/pagination), [Streaming](https://docs.apify.com/api/client/python/docs/concepts/streaming-resources)).
|
|
174
174
|
- **Convenience methods** — `call()`, `wait_for_finish()`, nested resource access, and other shortcuts that hide platform quirks ([Convenience methods](https://docs.apify.com/api/client/python/docs/concepts/convenience-methods)).
|
|
175
|
-
- **Pluggable HTTP layer** — use the default [Impit](https://github.com/apify/impit)-based client, opt in to the built-in [
|
|
175
|
+
- **Pluggable HTTP layer** — use the default [Impit](https://github.com/apify/impit)-based client, opt in to the built-in [HTTPX2](https://github.com/pydantic/httpx2) client, or plug in any custom implementation ([HTTP clients](https://docs.apify.com/api/client/python/docs/concepts/custom-http-clients)).
|
|
176
176
|
- **Structured errors** — every API error surfaces as an [`ApifyApiError`](https://docs.apify.com/api/client/python/reference/class/ApifyApiError) with HTTP-specific subclasses for precise handling ([Error handling](https://docs.apify.com/api/client/python/docs/concepts/error-handling)).
|
|
177
177
|
- **Debug logging** — opt-in structured logging on the `apify_client` logger captures request URLs, status codes, retry attempts, and more ([Logging](https://docs.apify.com/api/client/python/docs/concepts/logging)).
|
|
178
178
|
|
|
@@ -58,7 +58,7 @@
|
|
|
58
58
|
```
|
|
59
59
|
|
|
60
60
|
[Impit](https://github.com/apify/impit) is the default HTTP client and is installed automatically. To use the
|
|
61
|
-
built-in [
|
|
61
|
+
built-in [HTTPX2](https://github.com/pydantic/httpx2) client instead, install the optional `httpx2` extra, which
|
|
62
62
|
provides Pydantic's maintained continuation of HTTPX, and pass `http_client=Httpx2HttpClient()` to
|
|
63
63
|
`ApifyClient.with_custom_http_client()`:
|
|
64
64
|
|
|
@@ -135,7 +135,7 @@ For a guided walkthrough — authenticating, running an Actor, and reading its r
|
|
|
135
135
|
- **Tiered timeouts** — short / medium / long tiers picked per endpoint, overridable per call ([Timeouts](https://docs.apify.com/api/client/python/docs/concepts/timeouts)).
|
|
136
136
|
- **Pagination and streaming** — iterate datasets, key-value store keys, or live logs without manual paging or buffering ([Pagination](https://docs.apify.com/api/client/python/docs/concepts/pagination), [Streaming](https://docs.apify.com/api/client/python/docs/concepts/streaming-resources)).
|
|
137
137
|
- **Convenience methods** — `call()`, `wait_for_finish()`, nested resource access, and other shortcuts that hide platform quirks ([Convenience methods](https://docs.apify.com/api/client/python/docs/concepts/convenience-methods)).
|
|
138
|
-
- **Pluggable HTTP layer** — use the default [Impit](https://github.com/apify/impit)-based client, opt in to the built-in [
|
|
138
|
+
- **Pluggable HTTP layer** — use the default [Impit](https://github.com/apify/impit)-based client, opt in to the built-in [HTTPX2](https://github.com/pydantic/httpx2) client, or plug in any custom implementation ([HTTP clients](https://docs.apify.com/api/client/python/docs/concepts/custom-http-clients)).
|
|
139
139
|
- **Structured errors** — every API error surfaces as an [`ApifyApiError`](https://docs.apify.com/api/client/python/reference/class/ApifyApiError) with HTTP-specific subclasses for precise handling ([Error handling](https://docs.apify.com/api/client/python/docs/concepts/error-handling)).
|
|
140
140
|
- **Debug logging** — opt-in structured logging on the `apify_client` logger captures request URLs, status codes, retry attempts, and more ([Logging](https://docs.apify.com/api/client/python/docs/concepts/logging)).
|
|
141
141
|
|
|
@@ -19,9 +19,10 @@ The value of 1000 keeps backwards compatibility with the previous fixed cache si
|
|
|
19
19
|
class HasItems(Protocol[T]):
|
|
20
20
|
"""Structural contract for a single page of results from a paginated API endpoint.
|
|
21
21
|
|
|
22
|
-
Implementations must expose `items`. They may optionally expose `count` - the number of
|
|
23
|
-
this page, which
|
|
24
|
-
`count` opportunistically via `getattr` for offset bookkeeping and
|
|
22
|
+
Implementations must expose `items`. They may optionally expose `count` - the number of rows the API scanned to
|
|
23
|
+
produce this page, which `len(items)` can land below (filters drop items) or above (`unwind` splits one row into
|
|
24
|
+
several items). The iterator helpers consult `count` opportunistically via `getattr` for offset bookkeeping and
|
|
25
|
+
fall back to `len(items)` when it is absent.
|
|
25
26
|
"""
|
|
26
27
|
|
|
27
28
|
items: list[T]
|
|
@@ -38,20 +39,19 @@ def get_items_iterator(
|
|
|
38
39
|
|
|
39
40
|
The `callback` is invoked lazily to fetch each page from the API. It must accept `limit` and `offset` keyword
|
|
40
41
|
arguments and return an object whose `items` attribute is a list. If the object also exposes a `count` attribute, it
|
|
41
|
-
is used for offset bookkeeping
|
|
42
|
-
filters are applied).
|
|
42
|
+
is used for offset bookkeeping - `_page_scanned_rows` describes how the next offset is derived.
|
|
43
43
|
|
|
44
|
-
Iteration stops when a page scans no
|
|
45
|
-
|
|
46
|
-
|
|
47
|
-
|
|
48
|
-
between calls.
|
|
44
|
+
Iteration stops when a page scans no rows or when the user-requested `limit` is reached. A page can scan rows while
|
|
45
|
+
returning no items - filters like `clean` drop items from `items` but still count toward `count` - so terminating on
|
|
46
|
+
scanned rather than returned rows keeps the iterator advancing across fully-filtered pages. The `total` field is
|
|
47
|
+
intentionally not consulted, because it can change between calls.
|
|
49
48
|
|
|
50
49
|
Args:
|
|
51
50
|
callback: Function returning a single page of items.
|
|
52
|
-
limit: Maximum total number of
|
|
51
|
+
limit: Maximum total number of rows scanned across all pages. On the dataset items endpoint `unwind` can
|
|
52
|
+
turn one row into several items, so more items than this can be yielded. `None` or `0` means no limit.
|
|
53
53
|
offset: Starting offset for the first page.
|
|
54
|
-
chunk_size:
|
|
54
|
+
chunk_size: Per-page cap, sent to the API as its `limit`. `None` or `0` lets the API decide.
|
|
55
55
|
"""
|
|
56
56
|
effective_chunk = chunk_size or 0
|
|
57
57
|
initial_offset = offset or 0
|
|
@@ -59,13 +59,14 @@ def get_items_iterator(
|
|
|
59
59
|
fetched_items = 0
|
|
60
60
|
|
|
61
61
|
while True:
|
|
62
|
+
page_limit = _next_page_limit(initial_limit, fetched_items, effective_chunk)
|
|
62
63
|
current_page = callback(
|
|
63
|
-
limit=
|
|
64
|
+
limit=page_limit,
|
|
64
65
|
offset=initial_offset + fetched_items,
|
|
65
66
|
)
|
|
66
67
|
yield from current_page.items
|
|
67
68
|
|
|
68
|
-
page_scanned =
|
|
69
|
+
page_scanned = _page_scanned_rows(current_page, page_limit)
|
|
69
70
|
fetched_items += page_scanned
|
|
70
71
|
|
|
71
72
|
if not page_scanned or (initial_limit and fetched_items >= initial_limit):
|
|
@@ -89,14 +90,15 @@ async def get_items_iterator_async(
|
|
|
89
90
|
fetched_items = 0
|
|
90
91
|
|
|
91
92
|
while True:
|
|
93
|
+
page_limit = _next_page_limit(initial_limit, fetched_items, effective_chunk)
|
|
92
94
|
current_page = await callback(
|
|
93
|
-
limit=
|
|
95
|
+
limit=page_limit,
|
|
94
96
|
offset=initial_offset + fetched_items,
|
|
95
97
|
)
|
|
96
98
|
for item in current_page.items:
|
|
97
99
|
yield item
|
|
98
100
|
|
|
99
|
-
page_scanned =
|
|
101
|
+
page_scanned = _page_scanned_rows(current_page, page_limit)
|
|
100
102
|
fetched_items += page_scanned
|
|
101
103
|
|
|
102
104
|
if not page_scanned or (initial_limit and fetched_items >= initial_limit):
|
|
@@ -126,20 +128,30 @@ def get_cursor_iterator(
|
|
|
126
128
|
limit: int | None = None,
|
|
127
129
|
chunk_size: int | None = None,
|
|
128
130
|
) -> Iterator[KeyValueStoreKey] | Iterator[Request]:
|
|
129
|
-
"""Yield individual items from cursor-paginated API
|
|
131
|
+
"""Yield individual items from a cursor-paginated API response.
|
|
132
|
+
|
|
133
|
+
This iterator supports the two API responses that use cursor pagination. `ListOfKeys` is used for key-value store
|
|
134
|
+
keys, while `ListOfRequests` is used for request queue requests.
|
|
135
|
+
|
|
136
|
+
Pagination continues until either:
|
|
137
|
+
|
|
138
|
+
- the API returns no next cursor, or
|
|
139
|
+
- the requested `limit` is reached.
|
|
140
|
+
|
|
141
|
+
An empty page does not explicitly stop the iteration. In practice, both supported endpoints return a next cursor
|
|
142
|
+
only when the current page contains items, so an empty page always has a `None` cursor and naturally ends the
|
|
143
|
+
iteration.
|
|
130
144
|
|
|
131
|
-
|
|
132
|
-
|
|
133
|
-
|
|
134
|
-
|
|
135
|
-
cursor, not on whether a page returned items. Unlike offset responses, cursor responses expose no scanned-item
|
|
136
|
-
`count`, so `count` cannot be used to detect a fully-filtered page here.
|
|
145
|
+
The endpoints determine the next cursor differently:
|
|
146
|
+
|
|
147
|
+
- For key-value store keys, the cursor is the last key returned on the current page.
|
|
148
|
+
- For request queue requests, a cursor is returned only when the current page is full.
|
|
137
149
|
|
|
138
150
|
Args:
|
|
139
|
-
callback: Function
|
|
140
|
-
cursor:
|
|
151
|
+
callback: Function that returns one page of items and accepts `cursor` and `limit` keyword arguments.
|
|
152
|
+
cursor: Cursor to use for the first request. If `None`, iteration starts from the beginning.
|
|
141
153
|
limit: Maximum total number of items to yield across all pages.
|
|
142
|
-
chunk_size: Maximum number of items
|
|
154
|
+
chunk_size: Maximum number of items to request in a single API call.
|
|
143
155
|
"""
|
|
144
156
|
effective_chunk = chunk_size or 0
|
|
145
157
|
initial_limit = limit or 0
|
|
@@ -218,3 +230,19 @@ def _next_page_limit(initial_limit: int, fetched_items: int, effective_chunk: in
|
|
|
218
230
|
if not effective_chunk:
|
|
219
231
|
return remaining
|
|
220
232
|
return min(remaining, effective_chunk)
|
|
233
|
+
|
|
234
|
+
|
|
235
|
+
def _page_scanned_rows(page: HasItems[T], requested_limit: int) -> int:
|
|
236
|
+
"""Compute how far the offset advances past `page`, in dataset rows.
|
|
237
|
+
|
|
238
|
+
Neither reported number is right on its own. `count` follows the rows the API scanned, but it is derived from a
|
|
239
|
+
dataset's item count, which is incremented by a throttled write and so lags a fresh push. `len(items)` counts the
|
|
240
|
+
items the API shaped out of those rows: filters (`clean`, `skip_empty`, `skip_hidden`) drop some, and `unwind`
|
|
241
|
+
splits one row into several. The larger of the two absorbs a `count` that lags behind the items returned, and
|
|
242
|
+
capping it at the rows the call asked for keeps an unwound page from advancing past rows the next call would then
|
|
243
|
+
never read. The cap is a valid bound because the endpoint applies the `limit` it is sent verbatim; on a page
|
|
244
|
+
covering fewer rows than that, the advance can still overshoot into rows a concurrent push appends afterwards. A
|
|
245
|
+
`requested_limit` of `0` means the call sent no limit, leaving the advance unbounded.
|
|
246
|
+
"""
|
|
247
|
+
scanned_rows = max(getattr(page, 'count', 0), len(page.items))
|
|
248
|
+
return min(scanned_rows, requested_limit) if requested_limit else scanned_rows
|
|
@@ -40,7 +40,7 @@ class DatasetItemsPage:
|
|
|
40
40
|
"""The offset of the first item in this page."""
|
|
41
41
|
|
|
42
42
|
count: int
|
|
43
|
-
"""Number of
|
|
43
|
+
"""Number of dataset rows the API scanned for this page, or the number of items returned when that is larger."""
|
|
44
44
|
|
|
45
45
|
limit: int
|
|
46
46
|
"""The limit that was used for this request."""
|
|
@@ -204,8 +204,8 @@ class DatasetClient(ResourceClient):
|
|
|
204
204
|
items=items,
|
|
205
205
|
total=int(response.headers['x-apify-pagination-total']),
|
|
206
206
|
offset=int(response.headers['x-apify-pagination-offset']),
|
|
207
|
-
#
|
|
208
|
-
#
|
|
207
|
+
# The header counts the rows the API scanned, which `unwind` and a lagging dataset item count can
|
|
208
|
+
# both leave below the number of items returned.
|
|
209
209
|
count=max(int(response.headers['x-apify-pagination-count']), len(items)),
|
|
210
210
|
# API returns 999999999999 when no limit is used
|
|
211
211
|
limit=int(response.headers['x-apify-pagination-limit']),
|
|
@@ -237,7 +237,8 @@ class DatasetClient(ResourceClient):
|
|
|
237
237
|
|
|
238
238
|
Args:
|
|
239
239
|
offset: Number of items that should be skipped at the start. The default value is 0.
|
|
240
|
-
limit: Maximum number of
|
|
240
|
+
limit: Maximum number of dataset rows to scan. Fewer items are yielded when filters drop some, more
|
|
241
|
+
when `unwind` splits a row into several. By default there is no limit.
|
|
241
242
|
desc: By default, results are returned in the same order as they were stored. To reverse the order,
|
|
242
243
|
set this parameter to True.
|
|
243
244
|
clean: If True, returns only non-empty items and skips hidden fields (i.e. fields starting with
|
|
@@ -260,7 +261,7 @@ class DatasetClient(ResourceClient):
|
|
|
260
261
|
skip_hidden: If True, then hidden fields are skipped from the output, i.e. fields starting with
|
|
261
262
|
the # character.
|
|
262
263
|
signature: Signature used to access the items.
|
|
263
|
-
chunk_size: Maximum number of
|
|
264
|
+
chunk_size: Maximum number of dataset rows requested per API call when iterating across pages.
|
|
264
265
|
timeout: Timeout for the API HTTP request.
|
|
265
266
|
|
|
266
267
|
Yields:
|
|
@@ -763,8 +764,8 @@ class DatasetClientAsync(ResourceClientAsync):
|
|
|
763
764
|
items=items,
|
|
764
765
|
total=int(response.headers['x-apify-pagination-total']),
|
|
765
766
|
offset=int(response.headers['x-apify-pagination-offset']),
|
|
766
|
-
#
|
|
767
|
-
#
|
|
767
|
+
# The header counts the rows the API scanned, which `unwind` and a lagging dataset item count can
|
|
768
|
+
# both leave below the number of items returned.
|
|
768
769
|
count=max(int(response.headers['x-apify-pagination-count']), len(items)),
|
|
769
770
|
# API returns 999999999999 when no limit is used
|
|
770
771
|
limit=int(response.headers['x-apify-pagination-limit']),
|
|
@@ -796,7 +797,8 @@ class DatasetClientAsync(ResourceClientAsync):
|
|
|
796
797
|
|
|
797
798
|
Args:
|
|
798
799
|
offset: Number of items that should be skipped at the start. The default value is 0.
|
|
799
|
-
limit: Maximum number of
|
|
800
|
+
limit: Maximum number of dataset rows to scan. Fewer items are yielded when filters drop some, more
|
|
801
|
+
when `unwind` splits a row into several. By default there is no limit.
|
|
800
802
|
desc: By default, results are returned in the same order as they were stored. To reverse the order,
|
|
801
803
|
set this parameter to True.
|
|
802
804
|
clean: If True, returns only non-empty items and skips hidden fields (i.e. fields starting with
|
|
@@ -819,7 +821,7 @@ class DatasetClientAsync(ResourceClientAsync):
|
|
|
819
821
|
skip_hidden: If True, then hidden fields are skipped from the output, i.e. fields starting with
|
|
820
822
|
the # character.
|
|
821
823
|
signature: Signature used to access the items.
|
|
822
|
-
chunk_size: Maximum number of
|
|
824
|
+
chunk_size: Maximum number of dataset rows requested per API call when iterating across pages.
|
|
823
825
|
timeout: Timeout for the API HTTP request.
|
|
824
826
|
|
|
825
827
|
Yields:
|
|
@@ -7,7 +7,7 @@ import threading
|
|
|
7
7
|
from asyncio import Task
|
|
8
8
|
from datetime import UTC, datetime
|
|
9
9
|
from threading import Thread
|
|
10
|
-
from typing import TYPE_CHECKING, ClassVar, Self
|
|
10
|
+
from typing import TYPE_CHECKING, ClassVar, Self
|
|
11
11
|
|
|
12
12
|
from apify_client._docs import docs_group
|
|
13
13
|
|
|
@@ -79,10 +79,10 @@ class StreamedLogBase:
|
|
|
79
79
|
"""Guess the log level from the message."""
|
|
80
80
|
# Using only levels explicitly mentioned in the logging module
|
|
81
81
|
known_levels = ('CRITICAL', 'FATAL', 'ERROR', 'WARN', 'WARNING', 'INFO', 'DEBUG', 'NOTSET')
|
|
82
|
+
level_names_to_levels = logging.getLevelNamesMapping()
|
|
82
83
|
for level in known_levels:
|
|
83
84
|
if level in message:
|
|
84
|
-
|
|
85
|
-
return cast('int', logging.getLevelName(level))
|
|
85
|
+
return level_names_to_levels[level]
|
|
86
86
|
# Unknown log level. Fall back to the default.
|
|
87
87
|
return logging.INFO
|
|
88
88
|
|
|
@@ -5,7 +5,7 @@ from apify_client.http_clients._impit import ImpitHttpClient, ImpitHttpClientAsy
|
|
|
5
5
|
|
|
6
6
|
_install_import_hook(__name__)
|
|
7
7
|
|
|
8
|
-
# `httpx2` is an optional extra, so the import is wrapped in try_import. Accessing the
|
|
8
|
+
# `httpx2` is an optional extra, so the import is wrapped in try_import. Accessing the HTTPX2 clients without the
|
|
9
9
|
# extra installed raises a clear ImportError instead of failing at package import time.
|
|
10
10
|
with _try_import(
|
|
11
11
|
__name__,
|
|
@@ -2,7 +2,7 @@ from __future__ import annotations
|
|
|
2
2
|
|
|
3
3
|
from typing import TYPE_CHECKING
|
|
4
4
|
|
|
5
|
-
import httpx2
|
|
5
|
+
import httpx2
|
|
6
6
|
from typing_extensions import override
|
|
7
7
|
|
|
8
8
|
from apify_client._consts import (
|
|
@@ -24,32 +24,32 @@ if TYPE_CHECKING:
|
|
|
24
24
|
|
|
25
25
|
|
|
26
26
|
_PERMANENT_ERRORS = (
|
|
27
|
-
# A request
|
|
28
|
-
|
|
29
|
-
# A URL scheme
|
|
30
|
-
|
|
27
|
+
# A request HTTPX2 rejects before sending it, e.g. one carrying an invalid header value.
|
|
28
|
+
httpx2.LocalProtocolError,
|
|
29
|
+
# A URL scheme HTTPX2 refuses to speak, which repeating the request cannot change.
|
|
30
|
+
httpx2.UnsupportedProtocol,
|
|
31
31
|
# An over-long redirect chain is a routing loop, which repeating the request cannot break.
|
|
32
|
-
|
|
32
|
+
httpx2.TooManyRedirects,
|
|
33
33
|
# Only `Response.raise_for_status()` raises this, and the client never calls it - the shared pipeline decides on
|
|
34
34
|
# status codes from the response itself.
|
|
35
|
-
|
|
35
|
+
httpx2.HTTPStatusError,
|
|
36
36
|
)
|
|
37
|
-
"""
|
|
37
|
+
"""HTTPX2 errors that a retry cannot fix. Everything else in the `httpx2.HTTPError` tree counts as transient."""
|
|
38
38
|
|
|
39
39
|
|
|
40
40
|
@docs_group('HTTP clients')
|
|
41
41
|
class Httpx2HttpClient(HttpClient):
|
|
42
|
-
"""Synchronous HTTP client for the Apify API built on top of [
|
|
42
|
+
"""Synchronous HTTP client for the Apify API built on top of [HTTPX2](https://github.com/pydantic/httpx2).
|
|
43
43
|
|
|
44
|
-
This client wraps `
|
|
44
|
+
This client wraps `httpx2.Client` and adds automatic retries with exponential backoff for rate-limited
|
|
45
45
|
(HTTP 429) and server error (HTTP 5xx) responses.
|
|
46
46
|
|
|
47
|
-
|
|
47
|
+
HTTPX2 applies a request timeout to each socket operation rather than to the request as a whole, so a response
|
|
48
48
|
whose body arrives slowly keeps resetting it and can outlast both the requested timeout and `timeout_max`. The
|
|
49
49
|
default Impit client enforces the same value as a deadline for the whole request, body included.
|
|
50
50
|
|
|
51
51
|
Requires the `httpx2` extra: `pip install "apify-client[httpx2]"`. The `httpx2` package is Pydantic's maintained
|
|
52
|
-
continuation of HTTPX
|
|
52
|
+
continuation of HTTPX.
|
|
53
53
|
"""
|
|
54
54
|
|
|
55
55
|
def __init__(
|
|
@@ -66,7 +66,7 @@ class Httpx2HttpClient(HttpClient):
|
|
|
66
66
|
headers: dict[str, str] | None = None,
|
|
67
67
|
http_compressor: HttpCompressor | None = None,
|
|
68
68
|
) -> None:
|
|
69
|
-
"""Initialize the
|
|
69
|
+
"""Initialize the HTTPX2-based synchronous HTTP client.
|
|
70
70
|
|
|
71
71
|
Args:
|
|
72
72
|
token: Apify API token for authentication.
|
|
@@ -93,27 +93,27 @@ class Httpx2HttpClient(HttpClient):
|
|
|
93
93
|
http_compressor=http_compressor,
|
|
94
94
|
)
|
|
95
95
|
|
|
96
|
-
self.
|
|
96
|
+
self._httpx2_client = httpx2.Client(
|
|
97
97
|
follow_redirects=True,
|
|
98
98
|
event_hooks={'response': [self._clear_response_cookies]},
|
|
99
99
|
)
|
|
100
100
|
|
|
101
101
|
@override
|
|
102
102
|
def is_timeout_error(self, exc: Exception) -> bool:
|
|
103
|
-
return super().is_timeout_error(exc) or isinstance(exc,
|
|
103
|
+
return super().is_timeout_error(exc) or isinstance(exc, httpx2.TimeoutException)
|
|
104
104
|
|
|
105
105
|
@override
|
|
106
106
|
def is_retryable_transport_error(self, exc: Exception) -> bool:
|
|
107
|
-
# Every error from
|
|
108
|
-
# `_PERMANENT_ERRORS`. Retrying is the default so a subclass
|
|
107
|
+
# Every error from HTTPX2's own hierarchy counts as transient except the permanently-failing types listed in
|
|
108
|
+
# `_PERMANENT_ERRORS`. Retrying is the default so a subclass HTTPX2 adds later is retried rather than
|
|
109
109
|
# silently treated as fatal. HTTP status code errors are handled by the shared pipeline based on the
|
|
110
110
|
# response status code, not here.
|
|
111
|
-
return isinstance(exc,
|
|
111
|
+
return isinstance(exc, httpx2.HTTPError) and not isinstance(exc, _PERMANENT_ERRORS)
|
|
112
112
|
|
|
113
113
|
@override
|
|
114
114
|
def close(self) -> None:
|
|
115
|
-
"""Close the underlying
|
|
116
|
-
self.
|
|
115
|
+
"""Close the underlying HTTPX2 connection pool."""
|
|
116
|
+
self._httpx2_client.close()
|
|
117
117
|
|
|
118
118
|
@override
|
|
119
119
|
def send_request(
|
|
@@ -125,8 +125,8 @@ class Httpx2HttpClient(HttpClient):
|
|
|
125
125
|
content: bytes | None,
|
|
126
126
|
timeout: float | None,
|
|
127
127
|
stream: bool,
|
|
128
|
-
) ->
|
|
129
|
-
request = self.
|
|
128
|
+
) -> httpx2.Response:
|
|
129
|
+
request = self._httpx2_client.build_request(
|
|
130
130
|
method=method,
|
|
131
131
|
url=url,
|
|
132
132
|
headers=headers,
|
|
@@ -134,26 +134,26 @@ class Httpx2HttpClient(HttpClient):
|
|
|
134
134
|
timeout=timeout,
|
|
135
135
|
)
|
|
136
136
|
_restore_explicit_cookie_header(request, headers)
|
|
137
|
-
return self.
|
|
137
|
+
return self._httpx2_client.send(request, stream=stream)
|
|
138
138
|
|
|
139
|
-
def _clear_response_cookies(self, _response:
|
|
140
|
-
"""Prevent
|
|
141
|
-
self.
|
|
139
|
+
def _clear_response_cookies(self, _response: httpx2.Response) -> None:
|
|
140
|
+
"""Prevent HTTPX2's shared cookie jar from leaking server cookies into later API requests."""
|
|
141
|
+
self._httpx2_client.cookies.clear()
|
|
142
142
|
|
|
143
143
|
|
|
144
144
|
@docs_group('HTTP clients')
|
|
145
145
|
class Httpx2HttpClientAsync(HttpClientAsync):
|
|
146
|
-
"""Asynchronous HTTP client for the Apify API built on top of [
|
|
146
|
+
"""Asynchronous HTTP client for the Apify API built on top of [HTTPX2](https://github.com/pydantic/httpx2).
|
|
147
147
|
|
|
148
|
-
This client wraps `
|
|
148
|
+
This client wraps `httpx2.AsyncClient` and adds automatic retries with exponential backoff for rate-limited
|
|
149
149
|
(HTTP 429) and server error (HTTP 5xx) responses.
|
|
150
150
|
|
|
151
|
-
|
|
151
|
+
HTTPX2 applies a request timeout to each socket operation rather than to the request as a whole, so a response
|
|
152
152
|
whose body arrives slowly keeps resetting it and can outlast both the requested timeout and `timeout_max`. The
|
|
153
153
|
default Impit client enforces the same value as a deadline for the whole request, body included.
|
|
154
154
|
|
|
155
155
|
Requires the `httpx2` extra: `pip install "apify-client[httpx2]"`. The `httpx2` package is Pydantic's maintained
|
|
156
|
-
continuation of HTTPX
|
|
156
|
+
continuation of HTTPX.
|
|
157
157
|
"""
|
|
158
158
|
|
|
159
159
|
def __init__(
|
|
@@ -170,7 +170,7 @@ class Httpx2HttpClientAsync(HttpClientAsync):
|
|
|
170
170
|
headers: dict[str, str] | None = None,
|
|
171
171
|
http_compressor: HttpCompressor | None = None,
|
|
172
172
|
) -> None:
|
|
173
|
-
"""Initialize the
|
|
173
|
+
"""Initialize the HTTPX2-based asynchronous HTTP client.
|
|
174
174
|
|
|
175
175
|
Args:
|
|
176
176
|
token: Apify API token for authentication.
|
|
@@ -197,27 +197,27 @@ class Httpx2HttpClientAsync(HttpClientAsync):
|
|
|
197
197
|
http_compressor=http_compressor,
|
|
198
198
|
)
|
|
199
199
|
|
|
200
|
-
self.
|
|
200
|
+
self._httpx2_async_client = httpx2.AsyncClient(
|
|
201
201
|
follow_redirects=True,
|
|
202
202
|
event_hooks={'response': [self._clear_response_cookies]},
|
|
203
203
|
)
|
|
204
204
|
|
|
205
205
|
@override
|
|
206
206
|
def is_timeout_error(self, exc: Exception) -> bool:
|
|
207
|
-
return super().is_timeout_error(exc) or isinstance(exc,
|
|
207
|
+
return super().is_timeout_error(exc) or isinstance(exc, httpx2.TimeoutException)
|
|
208
208
|
|
|
209
209
|
@override
|
|
210
210
|
def is_retryable_transport_error(self, exc: Exception) -> bool:
|
|
211
|
-
# Every error from
|
|
212
|
-
# `_PERMANENT_ERRORS`. Retrying is the default so a subclass
|
|
211
|
+
# Every error from HTTPX2's own hierarchy counts as transient except the permanently-failing types listed in
|
|
212
|
+
# `_PERMANENT_ERRORS`. Retrying is the default so a subclass HTTPX2 adds later is retried rather than
|
|
213
213
|
# silently treated as fatal. HTTP status code errors are handled by the shared pipeline based on the
|
|
214
214
|
# response status code, not here.
|
|
215
|
-
return isinstance(exc,
|
|
215
|
+
return isinstance(exc, httpx2.HTTPError) and not isinstance(exc, _PERMANENT_ERRORS)
|
|
216
216
|
|
|
217
217
|
@override
|
|
218
218
|
async def aclose(self) -> None:
|
|
219
|
-
"""Close the underlying asynchronous
|
|
220
|
-
await self.
|
|
219
|
+
"""Close the underlying asynchronous HTTPX2 connection pool."""
|
|
220
|
+
await self._httpx2_async_client.aclose()
|
|
221
221
|
|
|
222
222
|
@override
|
|
223
223
|
async def send_request(
|
|
@@ -229,8 +229,8 @@ class Httpx2HttpClientAsync(HttpClientAsync):
|
|
|
229
229
|
content: bytes | None,
|
|
230
230
|
timeout: float | None,
|
|
231
231
|
stream: bool,
|
|
232
|
-
) ->
|
|
233
|
-
request = self.
|
|
232
|
+
) -> httpx2.Response:
|
|
233
|
+
request = self._httpx2_async_client.build_request(
|
|
234
234
|
method=method,
|
|
235
235
|
url=url,
|
|
236
236
|
headers=headers,
|
|
@@ -238,17 +238,17 @@ class Httpx2HttpClientAsync(HttpClientAsync):
|
|
|
238
238
|
timeout=timeout,
|
|
239
239
|
)
|
|
240
240
|
_restore_explicit_cookie_header(request, headers)
|
|
241
|
-
return await self.
|
|
241
|
+
return await self._httpx2_async_client.send(request, stream=stream)
|
|
242
242
|
|
|
243
|
-
async def _clear_response_cookies(self, _response:
|
|
244
|
-
"""Prevent
|
|
245
|
-
self.
|
|
243
|
+
async def _clear_response_cookies(self, _response: httpx2.Response) -> None:
|
|
244
|
+
"""Prevent HTTPX2's shared cookie jar from leaking server cookies into later API requests."""
|
|
245
|
+
self._httpx2_async_client.cookies.clear()
|
|
246
246
|
|
|
247
247
|
|
|
248
|
-
def _restore_explicit_cookie_header(request:
|
|
249
|
-
"""Keep only cookies explicitly supplied for this request, never cookies from
|
|
248
|
+
def _restore_explicit_cookie_header(request: httpx2.Request, headers: dict[str, str]) -> None:
|
|
249
|
+
"""Keep only cookies explicitly supplied for this request, never cookies from HTTPX2's shared jar.
|
|
250
250
|
|
|
251
|
-
|
|
251
|
+
HTTPX2 drops the `Cookie` header when it builds a redirect request and rebuilds it from the jar, so an explicit
|
|
252
252
|
cookie only reaches the first hop of a redirected request.
|
|
253
253
|
"""
|
|
254
254
|
explicit_cookie = next((value for key, value in headers.items() if key.lower() == 'cookie'), None)
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
{apify_client-3.2.0 → apify_client-3.2.1b2}/src/apify_client/_resource_clients/_resource_client.py
RENAMED
|
File without changes
|
|
File without changes
|
{apify_client-3.2.0 → apify_client-3.2.1b2}/src/apify_client/_resource_clients/actor_collection.py
RENAMED
|
File without changes
|
{apify_client-3.2.0 → apify_client-3.2.1b2}/src/apify_client/_resource_clients/actor_env_var.py
RENAMED
|
File without changes
|
|
File without changes
|
{apify_client-3.2.0 → apify_client-3.2.1b2}/src/apify_client/_resource_clients/actor_version.py
RENAMED
|
File without changes
|
|
File without changes
|
|
File without changes
|
{apify_client-3.2.0 → apify_client-3.2.1b2}/src/apify_client/_resource_clients/build_collection.py
RENAMED
|
File without changes
|
{apify_client-3.2.0 → apify_client-3.2.1b2}/src/apify_client/_resource_clients/dataset_collection.py
RENAMED
|
File without changes
|
{apify_client-3.2.0 → apify_client-3.2.1b2}/src/apify_client/_resource_clients/key_value_store.py
RENAMED
|
File without changes
|
|
File without changes
|
|
File without changes
|
{apify_client-3.2.0 → apify_client-3.2.1b2}/src/apify_client/_resource_clients/request_queue.py
RENAMED
|
File without changes
|
|
File without changes
|
|
File without changes
|
{apify_client-3.2.0 → apify_client-3.2.1b2}/src/apify_client/_resource_clients/run_collection.py
RENAMED
|
File without changes
|
|
File without changes
|
|
File without changes
|
{apify_client-3.2.0 → apify_client-3.2.1b2}/src/apify_client/_resource_clients/store_collection.py
RENAMED
|
File without changes
|
|
File without changes
|
{apify_client-3.2.0 → apify_client-3.2.1b2}/src/apify_client/_resource_clients/task_collection.py
RENAMED
|
File without changes
|
|
File without changes
|
|
File without changes
|
{apify_client-3.2.0 → apify_client-3.2.1b2}/src/apify_client/_resource_clients/webhook_collection.py
RENAMED
|
File without changes
|
{apify_client-3.2.0 → apify_client-3.2.1b2}/src/apify_client/_resource_clients/webhook_dispatch.py
RENAMED
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|