apify-client 3.2.2b1__tar.gz → 3.2.2b3__tar.gz
This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
- {apify_client-3.2.2b1 → apify_client-3.2.2b3}/CHANGELOG.md +6 -0
- {apify_client-3.2.2b1 → apify_client-3.2.2b3}/PKG-INFO +1 -1
- {apify_client-3.2.2b1 → apify_client-3.2.2b3}/pyproject.toml +2 -2
- {apify_client-3.2.2b1 → apify_client-3.2.2b3}/pyproject.toml.orig +2 -2
- {apify_client-3.2.2b1 → apify_client-3.2.2b3}/src/apify_client/_models.py +10 -0
- {apify_client-3.2.2b1 → apify_client-3.2.2b3}/src/apify_client/_resource_clients/_resource_client.py +11 -1
- {apify_client-3.2.2b1 → apify_client-3.2.2b3}/src/apify_client/_resource_clients/actor.py +65 -14
- {apify_client-3.2.2b1 → apify_client-3.2.2b3}/src/apify_client/_resource_clients/run.py +238 -1
- {apify_client-3.2.2b1 → apify_client-3.2.2b3}/src/apify_client/_resource_clients/task.py +57 -14
- apify_client-3.2.2b3/src/apify_client/_utils/wait_for_resources.py +82 -0
- {apify_client-3.2.2b1 → apify_client-3.2.2b3}/CONTRIBUTING.md +0 -0
- {apify_client-3.2.2b1 → apify_client-3.2.2b3}/LICENSE +0 -0
- {apify_client-3.2.2b1 → apify_client-3.2.2b3}/README.md +0 -0
- {apify_client-3.2.2b1 → apify_client-3.2.2b3}/src/apify_client/__init__.py +0 -0
- {apify_client-3.2.2b1 → apify_client-3.2.2b3}/src/apify_client/_apify_client.py +0 -0
- {apify_client-3.2.2b1 → apify_client-3.2.2b3}/src/apify_client/_client_registry.py +0 -0
- {apify_client-3.2.2b1 → apify_client-3.2.2b3}/src/apify_client/_consts.py +0 -0
- {apify_client-3.2.2b1 → apify_client-3.2.2b3}/src/apify_client/_docs.py +0 -0
- {apify_client-3.2.2b1 → apify_client-3.2.2b3}/src/apify_client/_literals.py +0 -0
- {apify_client-3.2.2b1 → apify_client-3.2.2b3}/src/apify_client/_logging.py +0 -0
- {apify_client-3.2.2b1 → apify_client-3.2.2b3}/src/apify_client/_pagination.py +0 -0
- {apify_client-3.2.2b1 → apify_client-3.2.2b3}/src/apify_client/_resource_clients/__init__.py +0 -0
- {apify_client-3.2.2b1 → apify_client-3.2.2b3}/src/apify_client/_resource_clients/actor_collection.py +0 -0
- {apify_client-3.2.2b1 → apify_client-3.2.2b3}/src/apify_client/_resource_clients/actor_env_var.py +0 -0
- {apify_client-3.2.2b1 → apify_client-3.2.2b3}/src/apify_client/_resource_clients/actor_env_var_collection.py +0 -0
- {apify_client-3.2.2b1 → apify_client-3.2.2b3}/src/apify_client/_resource_clients/actor_version.py +0 -0
- {apify_client-3.2.2b1 → apify_client-3.2.2b3}/src/apify_client/_resource_clients/actor_version_collection.py +0 -0
- {apify_client-3.2.2b1 → apify_client-3.2.2b3}/src/apify_client/_resource_clients/build.py +0 -0
- {apify_client-3.2.2b1 → apify_client-3.2.2b3}/src/apify_client/_resource_clients/build_collection.py +0 -0
- {apify_client-3.2.2b1 → apify_client-3.2.2b3}/src/apify_client/_resource_clients/dataset.py +0 -0
- {apify_client-3.2.2b1 → apify_client-3.2.2b3}/src/apify_client/_resource_clients/dataset_collection.py +0 -0
- {apify_client-3.2.2b1 → apify_client-3.2.2b3}/src/apify_client/_resource_clients/key_value_store.py +0 -0
- {apify_client-3.2.2b1 → apify_client-3.2.2b3}/src/apify_client/_resource_clients/key_value_store_collection.py +0 -0
- {apify_client-3.2.2b1 → apify_client-3.2.2b3}/src/apify_client/_resource_clients/log.py +0 -0
- {apify_client-3.2.2b1 → apify_client-3.2.2b3}/src/apify_client/_resource_clients/request_queue.py +0 -0
- {apify_client-3.2.2b1 → apify_client-3.2.2b3}/src/apify_client/_resource_clients/request_queue_collection.py +0 -0
- {apify_client-3.2.2b1 → apify_client-3.2.2b3}/src/apify_client/_resource_clients/run_collection.py +0 -0
- {apify_client-3.2.2b1 → apify_client-3.2.2b3}/src/apify_client/_resource_clients/schedule.py +0 -0
- {apify_client-3.2.2b1 → apify_client-3.2.2b3}/src/apify_client/_resource_clients/schedule_collection.py +0 -0
- {apify_client-3.2.2b1 → apify_client-3.2.2b3}/src/apify_client/_resource_clients/store_collection.py +0 -0
- {apify_client-3.2.2b1 → apify_client-3.2.2b3}/src/apify_client/_resource_clients/task_collection.py +0 -0
- {apify_client-3.2.2b1 → apify_client-3.2.2b3}/src/apify_client/_resource_clients/user.py +0 -0
- {apify_client-3.2.2b1 → apify_client-3.2.2b3}/src/apify_client/_resource_clients/webhook.py +0 -0
- {apify_client-3.2.2b1 → apify_client-3.2.2b3}/src/apify_client/_resource_clients/webhook_collection.py +0 -0
- {apify_client-3.2.2b1 → apify_client-3.2.2b3}/src/apify_client/_resource_clients/webhook_dispatch.py +0 -0
- {apify_client-3.2.2b1 → apify_client-3.2.2b3}/src/apify_client/_resource_clients/webhook_dispatch_collection.py +0 -0
- {apify_client-3.2.2b1 → apify_client-3.2.2b3}/src/apify_client/_statistics.py +0 -0
- {apify_client-3.2.2b1 → apify_client-3.2.2b3}/src/apify_client/_status_message_watcher.py +0 -0
- {apify_client-3.2.2b1 → apify_client-3.2.2b3}/src/apify_client/_streamed_log.py +0 -0
- {apify_client-3.2.2b1 → apify_client-3.2.2b3}/src/apify_client/_typeddicts.py +0 -0
- {apify_client-3.2.2b1 → apify_client-3.2.2b3}/src/apify_client/_utils/__init__.py +0 -0
- {apify_client-3.2.2b1 → apify_client-3.2.2b3}/src/apify_client/_utils/crypto.py +0 -0
- {apify_client-3.2.2b1 → apify_client-3.2.2b3}/src/apify_client/_utils/encoding.py +0 -0
- {apify_client-3.2.2b1 → apify_client-3.2.2b3}/src/apify_client/_utils/errors.py +0 -0
- {apify_client-3.2.2b1 → apify_client-3.2.2b3}/src/apify_client/_utils/http.py +0 -0
- {apify_client-3.2.2b1 → apify_client-3.2.2b3}/src/apify_client/_utils/time.py +0 -0
- {apify_client-3.2.2b1 → apify_client-3.2.2b3}/src/apify_client/_utils/try_import.py +0 -0
- {apify_client-3.2.2b1 → apify_client-3.2.2b3}/src/apify_client/errors.py +0 -0
- {apify_client-3.2.2b1 → apify_client-3.2.2b3}/src/apify_client/http_clients/__init__.py +0 -0
- {apify_client-3.2.2b1 → apify_client-3.2.2b3}/src/apify_client/http_clients/_base.py +0 -0
- {apify_client-3.2.2b1 → apify_client-3.2.2b3}/src/apify_client/http_clients/_httpx2.py +0 -0
- {apify_client-3.2.2b1 → apify_client-3.2.2b3}/src/apify_client/http_clients/_impit.py +0 -0
- {apify_client-3.2.2b1 → apify_client-3.2.2b3}/src/apify_client/http_clients/_streamed_body.py +0 -0
- {apify_client-3.2.2b1 → apify_client-3.2.2b3}/src/apify_client/http_compressors/__init__.py +0 -0
- {apify_client-3.2.2b1 → apify_client-3.2.2b3}/src/apify_client/http_compressors/_base.py +0 -0
- {apify_client-3.2.2b1 → apify_client-3.2.2b3}/src/apify_client/http_compressors/_brotli.py +0 -0
- {apify_client-3.2.2b1 → apify_client-3.2.2b3}/src/apify_client/http_compressors/_gzip.py +0 -0
- {apify_client-3.2.2b1 → apify_client-3.2.2b3}/src/apify_client/http_compressors/_resolve.py +0 -0
- {apify_client-3.2.2b1 → apify_client-3.2.2b3}/src/apify_client/py.typed +0 -0
- {apify_client-3.2.2b1 → apify_client-3.2.2b3}/src/apify_client/types.py +0 -0
|
@@ -8,6 +8,12 @@ All notable changes to this project will be documented in this file.
|
|
|
8
8
|
### 🚀 Features
|
|
9
9
|
|
|
10
10
|
- Stream request bodies from files, iterables, and responses ([#1060](https://github.com/apify/apify-client-python/pull/1060)) ([9be215b](https://github.com/apify/apify-client-python/commit/9be215b23a13de0abc4a255a18450cc764e7a6c7)) by [@vdusek](https://github.com/vdusek), closes [#972](https://github.com/apify/apify-client-python/issues/972)
|
|
11
|
+
- Retry starting a run on memory and concurrent-runs limits ([#1081](https://github.com/apify/apify-client-python/pull/1081)) ([3dd8ca5](https://github.com/apify/apify-client-python/commit/3dd8ca5fe4d2537cb009cb310ba12b39927a9558)) by [@vdusek](https://github.com/vdusek), closes [#1071](https://github.com/apify/apify-client-python/issues/1071)
|
|
12
|
+
- Add live iteration over a run's dataset items ([#1079](https://github.com/apify/apify-client-python/pull/1079)) ([6aa8ead](https://github.com/apify/apify-client-python/commit/6aa8ead676a3520f6a418597eb62f319a3d19d30)) by [@vdusek](https://github.com/vdusek), closes [#1065](https://github.com/apify/apify-client-python/issues/1065)
|
|
13
|
+
|
|
14
|
+
### 🐛 Bug Fixes
|
|
15
|
+
|
|
16
|
+
- Add missing `readme` field to `Profile` model ([#1086](https://github.com/apify/apify-client-python/pull/1086)) ([ae17148](https://github.com/apify/apify-client-python/commit/ae1714891c4f70d66ab4964c1783fee0a96776f1)) by [@apify-service-account](https://github.com/apify-service-account)
|
|
11
17
|
|
|
12
18
|
|
|
13
19
|
<!-- git-cliff-unreleased-end -->
|
|
@@ -4,7 +4,7 @@ build-backend = "uv_build"
|
|
|
4
4
|
|
|
5
5
|
[project]
|
|
6
6
|
name = "apify_client"
|
|
7
|
-
version = "3.2.
|
|
7
|
+
version = "3.2.2b3"
|
|
8
8
|
description = "Apify API client for Python"
|
|
9
9
|
license = "Apache-2.0"
|
|
10
10
|
license-files = ["LICENSE"]
|
|
@@ -246,7 +246,7 @@ disable_timestamp = true
|
|
|
246
246
|
keep_model_order = true
|
|
247
247
|
|
|
248
248
|
[tool.apify.openapi-spec]
|
|
249
|
-
version = "v2-2026-
|
|
249
|
+
version = "v2-2026-10-01T153946Z"
|
|
250
250
|
|
|
251
251
|
[tool.uv]
|
|
252
252
|
exclude-newer = "24 hours"
|
|
@@ -4,7 +4,7 @@ build-backend = "uv_build"
|
|
|
4
4
|
|
|
5
5
|
[project]
|
|
6
6
|
name = "apify_client"
|
|
7
|
-
version = "3.2.
|
|
7
|
+
version = "3.2.2b3"
|
|
8
8
|
description = "Apify API client for Python"
|
|
9
9
|
authors = [{ name = "Apify Technologies s.r.o.", email = "support@apify.com" }]
|
|
10
10
|
license = "Apache-2.0"
|
|
@@ -250,7 +250,7 @@ keep_model_order = true
|
|
|
250
250
|
# workflow committing nothing unless the models changed. A local regeneration that only moves this line is that
|
|
251
251
|
# same case - drop it rather than committing it alone.
|
|
252
252
|
[tool.apify.openapi-spec]
|
|
253
|
-
version = "v2-2026-
|
|
253
|
+
version = "v2-2026-10-01T153946Z"
|
|
254
254
|
|
|
255
255
|
[tool.uv]
|
|
256
256
|
# Minimal defense against supply-chain atatcks.
|
|
@@ -595,6 +595,12 @@ class Build(BaseModel):
|
|
|
595
595
|
build_number: Annotated[
|
|
596
596
|
str, Field(examples=['0.1.1'], pattern='^([0-9]|[1-9][0-9])\\.([0-9]|[1-9][0-9])(\\.[1-9][0-9]{0,4})$')
|
|
597
597
|
]
|
|
598
|
+
image_digest: Annotated[
|
|
599
|
+
str | None, Field(examples=['1b2f1e8c0d5a4c7f9e3b6a2d8c4e0f7a5b9d3c1e6f8a2b4d0c7e9f1a3b5d7c9e'])
|
|
600
|
+
] = None
|
|
601
|
+
"""
|
|
602
|
+
Digest of the built Docker image manifest, without the `sha256:` prefix. Compare digests of two builds to find out whether their image contents differ. `null` if the digest is not available.
|
|
603
|
+
"""
|
|
598
604
|
act_version: Annotated[ActVersion | None, Field(title='BuildActVersion')] = None
|
|
599
605
|
"""
|
|
600
606
|
Snapshot of the Actor version that this build was created from.
|
|
@@ -2280,6 +2286,10 @@ class Profile(BaseModel):
|
|
|
2280
2286
|
alias_generator=to_camel,
|
|
2281
2287
|
)
|
|
2282
2288
|
bio: Annotated[str | None, Field(examples=['I started web scraping in 1985 using Altair BASIC.'])] = None
|
|
2289
|
+
readme: Annotated[str | None, Field(examples=['### Hello world 👋🏻\nI build web scrapers.'])] = None
|
|
2290
|
+
"""
|
|
2291
|
+
Markdown README shown on the user's public profile page.
|
|
2292
|
+
"""
|
|
2283
2293
|
name: Annotated[str | None, Field(examples=['Jane Doe'])] = None
|
|
2284
2294
|
picture_url: Annotated[AnyUrl | None, Field(examples=['https://apify.com/img/anonymous_user_picture.png'])] = None
|
|
2285
2295
|
github_username: Annotated[str | None, Field(examples=['torvalds.'])] = None
|
{apify_client-3.2.2b1 → apify_client-3.2.2b3}/src/apify_client/_resource_clients/_resource_client.py
RENAMED
|
@@ -42,6 +42,7 @@ class ResourceClientBase(metaclass=WithLogDetailsClient):
|
|
|
42
42
|
client_registry: Any,
|
|
43
43
|
resource_id: str | None = None,
|
|
44
44
|
params: dict | None = None,
|
|
45
|
+
api_base_url: str | None = None,
|
|
45
46
|
) -> None:
|
|
46
47
|
"""Initialize the resource client.
|
|
47
48
|
|
|
@@ -53,11 +54,13 @@ class ResourceClientBase(metaclass=WithLogDetailsClient):
|
|
|
53
54
|
client_registry: Bundle of client classes for dependency injection.
|
|
54
55
|
resource_id: Optional resource ID for single-resource clients.
|
|
55
56
|
params: Optional default parameters for all requests.
|
|
57
|
+
api_base_url: Base URL of the API itself, for clients of top-level resources. Defaults to `base_url`.
|
|
56
58
|
"""
|
|
57
59
|
if resource_path.endswith('/'):
|
|
58
60
|
raise ValueError('resource_path must not end with "/"')
|
|
59
61
|
|
|
60
62
|
self._base_url = base_url
|
|
63
|
+
self._api_base_url = api_base_url or base_url
|
|
61
64
|
self._public_base_url = public_base_url
|
|
62
65
|
self._http_client = http_client
|
|
63
66
|
self._default_params = params or {}
|
|
@@ -82,11 +85,12 @@ class ResourceClientBase(metaclass=WithLogDetailsClient):
|
|
|
82
85
|
def _base_client_kwargs(self) -> dict[str, Any]:
|
|
83
86
|
"""Base kwargs for creating nested/child clients.
|
|
84
87
|
|
|
85
|
-
Returns dict with base_url, public_base_url, http_client, and client_registry. Caller adds
|
|
88
|
+
Returns dict with base_url, api_base_url, public_base_url, http_client, and client_registry. Caller adds
|
|
86
89
|
resource_path, resource_id, and params as needed.
|
|
87
90
|
"""
|
|
88
91
|
return {
|
|
89
92
|
'base_url': self._resource_url,
|
|
93
|
+
'api_base_url': self._api_base_url,
|
|
90
94
|
'public_base_url': self._public_base_url,
|
|
91
95
|
'http_client': self._http_client,
|
|
92
96
|
'client_registry': self._client_registry,
|
|
@@ -197,6 +201,7 @@ class ResourceClient(ResourceClientBase):
|
|
|
197
201
|
client_registry: ClientRegistry,
|
|
198
202
|
resource_id: str | None = None,
|
|
199
203
|
params: dict | None = None,
|
|
204
|
+
api_base_url: str | None = None,
|
|
200
205
|
) -> None:
|
|
201
206
|
"""Initialize the resource client.
|
|
202
207
|
|
|
@@ -208,6 +213,7 @@ class ResourceClient(ResourceClientBase):
|
|
|
208
213
|
client_registry: Bundle of client classes for dependency injection.
|
|
209
214
|
resource_id: Optional resource ID for single-resource clients.
|
|
210
215
|
params: Optional default parameters for all requests.
|
|
216
|
+
api_base_url: Base URL of the API itself, for clients of top-level resources. Defaults to `base_url`.
|
|
211
217
|
"""
|
|
212
218
|
super().__init__(
|
|
213
219
|
base_url=base_url,
|
|
@@ -217,6 +223,7 @@ class ResourceClient(ResourceClientBase):
|
|
|
217
223
|
client_registry=client_registry,
|
|
218
224
|
resource_id=resource_id,
|
|
219
225
|
params=params,
|
|
226
|
+
api_base_url=api_base_url,
|
|
220
227
|
)
|
|
221
228
|
|
|
222
229
|
def _get(self, *, timeout: Timeout) -> dict | None:
|
|
@@ -389,6 +396,7 @@ class ResourceClientAsync(ResourceClientBase):
|
|
|
389
396
|
client_registry: ClientRegistryAsync,
|
|
390
397
|
resource_id: str | None = None,
|
|
391
398
|
params: dict | None = None,
|
|
399
|
+
api_base_url: str | None = None,
|
|
392
400
|
) -> None:
|
|
393
401
|
"""Initialize the resource client.
|
|
394
402
|
|
|
@@ -400,6 +408,7 @@ class ResourceClientAsync(ResourceClientBase):
|
|
|
400
408
|
client_registry: Bundle of client classes for dependency injection.
|
|
401
409
|
resource_id: Optional resource ID for single-resource clients.
|
|
402
410
|
params: Optional default parameters for all requests.
|
|
411
|
+
api_base_url: Base URL of the API itself, for clients of top-level resources. Defaults to `base_url`.
|
|
403
412
|
"""
|
|
404
413
|
super().__init__(
|
|
405
414
|
base_url=base_url,
|
|
@@ -409,6 +418,7 @@ class ResourceClientAsync(ResourceClientBase):
|
|
|
409
418
|
client_registry=client_registry,
|
|
410
419
|
resource_id=resource_id,
|
|
411
420
|
params=params,
|
|
421
|
+
api_base_url=api_base_url,
|
|
412
422
|
)
|
|
413
423
|
|
|
414
424
|
async def _get(self, *, timeout: Timeout) -> dict | None:
|
|
@@ -26,6 +26,11 @@ from apify_client._resource_clients._resource_client import ResourceClient, Reso
|
|
|
26
26
|
from apify_client._utils.encoding import encode_key_value_store_record_value, encode_webhooks_to_base64
|
|
27
27
|
from apify_client._utils.http import response_to_dict
|
|
28
28
|
from apify_client._utils.time import to_seconds
|
|
29
|
+
from apify_client._utils.wait_for_resources import (
|
|
30
|
+
prepare_resendable_body,
|
|
31
|
+
start_waiting_for_resources,
|
|
32
|
+
start_waiting_for_resources_async,
|
|
33
|
+
)
|
|
29
34
|
|
|
30
35
|
if TYPE_CHECKING:
|
|
31
36
|
from datetime import timedelta
|
|
@@ -229,6 +234,7 @@ class ActorClient(ResourceClient):
|
|
|
229
234
|
force_permission_level: ActorPermissionLevel | None = None,
|
|
230
235
|
wait_for_finish: int | None = None,
|
|
231
236
|
webhooks: WebhooksList | None = None,
|
|
237
|
+
wait_for_resources: bool | timedelta = False,
|
|
232
238
|
timeout: Timeout = 'medium',
|
|
233
239
|
) -> Run:
|
|
234
240
|
"""Start the Actor and immediately return the Run object.
|
|
@@ -263,12 +269,21 @@ class ActorClient(ResourceClient):
|
|
|
263
269
|
* `event_types`: List of `WebhookEventType` values which trigger the webhook.
|
|
264
270
|
* `request_url`: URL to which to send the webhook HTTP request.
|
|
265
271
|
* `payload_template`: Optional template for the request payload.
|
|
272
|
+
wait_for_resources: Retry the start while the account lacks the memory or a concurrent-run slot for the run,
|
|
273
|
+
that is while the API rejects it with an `ApifyApiError` of type `actor-memory-limit-exceeded` or
|
|
274
|
+
`concurrent-runs-limit-exceeded`. Both clear as other runs or builds finish. The start is retried
|
|
275
|
+
every 10 seconds, and any other error is raised right away. `True` retries until the run starts, a
|
|
276
|
+
`timedelta` stops retrying after that long and raises the last error. A run that requests more memory
|
|
277
|
+
than the whole memory limit of the account is rejected with `actor-memory-limit-exceeded` as well and
|
|
278
|
+
never starts, so `True` retries it forever. A streamed `run_input` that cannot be rewound, such as a
|
|
279
|
+
generator, is sent only once, so its start is not retried.
|
|
266
280
|
timeout: Timeout for the API HTTP request.
|
|
267
281
|
|
|
268
282
|
Returns:
|
|
269
283
|
The run object.
|
|
270
284
|
"""
|
|
271
285
|
run_input, content_type = encode_key_value_store_record_value(run_input, content_type=content_type)
|
|
286
|
+
run_input, wait_for_resources = prepare_resendable_body(run_input, wait_for_resources=wait_for_resources)
|
|
272
287
|
|
|
273
288
|
request_params = self._build_params(
|
|
274
289
|
build=build,
|
|
@@ -282,13 +297,16 @@ class ActorClient(ResourceClient):
|
|
|
282
297
|
webhooks=encode_webhooks_to_base64(webhooks),
|
|
283
298
|
)
|
|
284
299
|
|
|
285
|
-
response =
|
|
286
|
-
|
|
287
|
-
|
|
288
|
-
|
|
289
|
-
|
|
290
|
-
|
|
291
|
-
|
|
300
|
+
response = start_waiting_for_resources(
|
|
301
|
+
lambda: self._http_client.call(
|
|
302
|
+
url=self._build_url('runs'),
|
|
303
|
+
method='POST',
|
|
304
|
+
headers={'content-type': content_type},
|
|
305
|
+
data=run_input,
|
|
306
|
+
params=request_params,
|
|
307
|
+
timeout=timeout,
|
|
308
|
+
),
|
|
309
|
+
wait_for_resources=wait_for_resources,
|
|
292
310
|
)
|
|
293
311
|
|
|
294
312
|
result = response_to_dict(response)
|
|
@@ -308,6 +326,7 @@ class ActorClient(ResourceClient):
|
|
|
308
326
|
webhooks: WebhooksList | None = None,
|
|
309
327
|
force_permission_level: ActorPermissionLevel | None = None,
|
|
310
328
|
wait_duration: timedelta | None = None,
|
|
329
|
+
wait_for_resources: bool | timedelta = False,
|
|
311
330
|
logger: Logger | Literal['default'] | None = 'default',
|
|
312
331
|
timeout: Timeout = 'no_timeout',
|
|
313
332
|
) -> Run | None:
|
|
@@ -341,6 +360,14 @@ class ActorClient(ResourceClient):
|
|
|
341
360
|
a webhook set up for the Actor, you do not have to add it again here.
|
|
342
361
|
wait_duration: The maximum time the server waits for the run to finish. If not provided,
|
|
343
362
|
waits indefinitely.
|
|
363
|
+
wait_for_resources: Retry the start while the account lacks the memory or a concurrent-run slot for the run,
|
|
364
|
+
that is while the API rejects it with an `ApifyApiError` of type `actor-memory-limit-exceeded` or
|
|
365
|
+
`concurrent-runs-limit-exceeded`. Both clear as other runs or builds finish. The start is retried
|
|
366
|
+
every 10 seconds, and any other error is raised right away. `True` retries until the run starts, a
|
|
367
|
+
`timedelta` stops retrying after that long and raises the last error. A run that requests more memory
|
|
368
|
+
than the whole memory limit of the account is rejected with `actor-memory-limit-exceeded` as well and
|
|
369
|
+
never starts, so `True` retries it forever. The time spent retrying doesn't count toward
|
|
370
|
+
`wait_duration`.
|
|
344
371
|
logger: Logger used to redirect logs from the Actor run. Using "default" literal means that a predefined
|
|
345
372
|
default logger will be used. Setting `None` will disable any log propagation. Passing custom logger
|
|
346
373
|
will redirect logs to the provided logger. The logger is also used to capture status and status message
|
|
@@ -361,6 +388,7 @@ class ActorClient(ResourceClient):
|
|
|
361
388
|
run_timeout=run_timeout,
|
|
362
389
|
webhooks=webhooks,
|
|
363
390
|
force_permission_level=force_permission_level,
|
|
391
|
+
wait_for_resources=wait_for_resources,
|
|
364
392
|
timeout=timeout,
|
|
365
393
|
)
|
|
366
394
|
run_client = self._client_registry.run_client(
|
|
@@ -740,6 +768,7 @@ class ActorClientAsync(ResourceClientAsync):
|
|
|
740
768
|
force_permission_level: ActorPermissionLevel | None = None,
|
|
741
769
|
wait_for_finish: int | None = None,
|
|
742
770
|
webhooks: WebhooksList | None = None,
|
|
771
|
+
wait_for_resources: bool | timedelta = False,
|
|
743
772
|
timeout: Timeout = 'medium',
|
|
744
773
|
) -> Run:
|
|
745
774
|
"""Start the Actor and immediately return the Run object.
|
|
@@ -774,12 +803,21 @@ class ActorClientAsync(ResourceClientAsync):
|
|
|
774
803
|
* `event_types`: List of `WebhookEventType` values which trigger the webhook.
|
|
775
804
|
* `request_url`: URL to which to send the webhook HTTP request.
|
|
776
805
|
* `payload_template`: Optional template for the request payload.
|
|
806
|
+
wait_for_resources: Retry the start while the account lacks the memory or a concurrent-run slot for the run,
|
|
807
|
+
that is while the API rejects it with an `ApifyApiError` of type `actor-memory-limit-exceeded` or
|
|
808
|
+
`concurrent-runs-limit-exceeded`. Both clear as other runs or builds finish. The start is retried
|
|
809
|
+
every 10 seconds, and any other error is raised right away. `True` retries until the run starts, a
|
|
810
|
+
`timedelta` stops retrying after that long and raises the last error. A run that requests more memory
|
|
811
|
+
than the whole memory limit of the account is rejected with `actor-memory-limit-exceeded` as well and
|
|
812
|
+
never starts, so `True` retries it forever. A streamed `run_input` that cannot be rewound, such as a
|
|
813
|
+
generator, is sent only once, so its start is not retried.
|
|
777
814
|
timeout: Timeout for the API HTTP request.
|
|
778
815
|
|
|
779
816
|
Returns:
|
|
780
817
|
The run object.
|
|
781
818
|
"""
|
|
782
819
|
run_input, content_type = encode_key_value_store_record_value(run_input, content_type=content_type)
|
|
820
|
+
run_input, wait_for_resources = prepare_resendable_body(run_input, wait_for_resources=wait_for_resources)
|
|
783
821
|
|
|
784
822
|
request_params = self._build_params(
|
|
785
823
|
build=build,
|
|
@@ -793,13 +831,16 @@ class ActorClientAsync(ResourceClientAsync):
|
|
|
793
831
|
webhooks=encode_webhooks_to_base64(webhooks),
|
|
794
832
|
)
|
|
795
833
|
|
|
796
|
-
response = await
|
|
797
|
-
|
|
798
|
-
|
|
799
|
-
|
|
800
|
-
|
|
801
|
-
|
|
802
|
-
|
|
834
|
+
response = await start_waiting_for_resources_async(
|
|
835
|
+
lambda: self._http_client.call(
|
|
836
|
+
url=self._build_url('runs'),
|
|
837
|
+
method='POST',
|
|
838
|
+
headers={'content-type': content_type},
|
|
839
|
+
data=run_input,
|
|
840
|
+
params=request_params,
|
|
841
|
+
timeout=timeout,
|
|
842
|
+
),
|
|
843
|
+
wait_for_resources=wait_for_resources,
|
|
803
844
|
)
|
|
804
845
|
|
|
805
846
|
result = response_to_dict(response)
|
|
@@ -819,6 +860,7 @@ class ActorClientAsync(ResourceClientAsync):
|
|
|
819
860
|
webhooks: WebhooksList | None = None,
|
|
820
861
|
force_permission_level: ActorPermissionLevel | None = None,
|
|
821
862
|
wait_duration: timedelta | None = None,
|
|
863
|
+
wait_for_resources: bool | timedelta = False,
|
|
822
864
|
logger: Logger | Literal['default'] | None = 'default',
|
|
823
865
|
timeout: Timeout = 'no_timeout',
|
|
824
866
|
) -> Run | None:
|
|
@@ -852,6 +894,14 @@ class ActorClientAsync(ResourceClientAsync):
|
|
|
852
894
|
a webhook set up for the Actor, you do not have to add it again here.
|
|
853
895
|
wait_duration: The maximum time the server waits for the run to finish. If not provided,
|
|
854
896
|
waits indefinitely.
|
|
897
|
+
wait_for_resources: Retry the start while the account lacks the memory or a concurrent-run slot for the run,
|
|
898
|
+
that is while the API rejects it with an `ApifyApiError` of type `actor-memory-limit-exceeded` or
|
|
899
|
+
`concurrent-runs-limit-exceeded`. Both clear as other runs or builds finish. The start is retried
|
|
900
|
+
every 10 seconds, and any other error is raised right away. `True` retries until the run starts, a
|
|
901
|
+
`timedelta` stops retrying after that long and raises the last error. A run that requests more memory
|
|
902
|
+
than the whole memory limit of the account is rejected with `actor-memory-limit-exceeded` as well and
|
|
903
|
+
never starts, so `True` retries it forever. The time spent retrying doesn't count toward
|
|
904
|
+
`wait_duration`.
|
|
855
905
|
logger: Logger used to redirect logs from the Actor run. Using "default" literal means that a predefined
|
|
856
906
|
default logger will be used. Setting `None` will disable any log propagation. Passing custom logger
|
|
857
907
|
will redirect logs to the provided logger. The logger is also used to capture status and status message
|
|
@@ -872,6 +922,7 @@ class ActorClientAsync(ResourceClientAsync):
|
|
|
872
922
|
run_timeout=run_timeout,
|
|
873
923
|
webhooks=webhooks,
|
|
874
924
|
force_permission_level=force_permission_level,
|
|
925
|
+
wait_for_resources=wait_for_resources,
|
|
875
926
|
timeout=timeout,
|
|
876
927
|
)
|
|
877
928
|
|
|
@@ -10,7 +10,8 @@ from typing import TYPE_CHECKING, Any
|
|
|
10
10
|
from apify_client._docs import docs_group
|
|
11
11
|
from apify_client._logging import create_redirect_logger
|
|
12
12
|
from apify_client._models import Run, RunResponse
|
|
13
|
-
from apify_client.
|
|
13
|
+
from apify_client._pagination import DEFAULT_CHUNK_SIZE
|
|
14
|
+
from apify_client._resource_clients._resource_client import _TERMINAL_STATUSES, ResourceClient, ResourceClientAsync
|
|
14
15
|
from apify_client._status_message_watcher import StatusMessageWatcher, StatusMessageWatcherAsync
|
|
15
16
|
from apify_client._streamed_log import StreamedLog, StreamedLogAsync
|
|
16
17
|
from apify_client._utils.encoding import encode_key_value_store_record_value
|
|
@@ -19,6 +20,7 @@ from apify_client._utils.time import to_seconds
|
|
|
19
20
|
|
|
20
21
|
if TYPE_CHECKING:
|
|
21
22
|
import logging
|
|
23
|
+
from collections.abc import AsyncIterator, Iterator
|
|
22
24
|
from decimal import Decimal
|
|
23
25
|
|
|
24
26
|
from apify_client._literals import GeneralAccess
|
|
@@ -32,6 +34,7 @@ if TYPE_CHECKING:
|
|
|
32
34
|
RequestQueueClient,
|
|
33
35
|
RequestQueueClientAsync,
|
|
34
36
|
)
|
|
37
|
+
from apify_client._resource_clients.dataset import DatasetItemsPage
|
|
35
38
|
from apify_client.types import Timeout
|
|
36
39
|
|
|
37
40
|
|
|
@@ -469,6 +472,122 @@ class RunClient(ResourceClient):
|
|
|
469
472
|
|
|
470
473
|
return StatusMessageWatcher(run_client=self, to_logger=to_logger, check_period=check_period)
|
|
471
474
|
|
|
475
|
+
def iterate_dataset_items(
|
|
476
|
+
self,
|
|
477
|
+
*,
|
|
478
|
+
offset: int | None = None,
|
|
479
|
+
limit: int | None = None,
|
|
480
|
+
clean: bool | None = None,
|
|
481
|
+
fields: list[str] | None = None,
|
|
482
|
+
omit: list[str] | None = None,
|
|
483
|
+
unwind: list[str] | None = None,
|
|
484
|
+
skip_empty: bool | None = None,
|
|
485
|
+
skip_hidden: bool | None = None,
|
|
486
|
+
chunk_size: int | None = None,
|
|
487
|
+
poll_interval: timedelta = timedelta(seconds=5),
|
|
488
|
+
timeout: Timeout = 'long',
|
|
489
|
+
) -> Iterator[dict]:
|
|
490
|
+
"""Iterate over the items of the run's default dataset while the run is still producing them.
|
|
491
|
+
|
|
492
|
+
While the run has not finished, each poll yields the rows below the dataset's `item_count` and then waits up to
|
|
493
|
+
`poll_interval` for the run to finish, so the last rows are read as soon as it does. Each page is requested with
|
|
494
|
+
a `limit` that ends at `item_count`, so it covers exactly the rows it asks for, whatever the filters or `unwind`
|
|
495
|
+
do to the items. `item_count` lags a few seconds behind the pushed items, so once the run reaches a terminal
|
|
496
|
+
status, the rows past it are read a page at a time until none are left, and the iterator returns. On a
|
|
497
|
+
`last_run()` client, the iterator sticks to the run that its first request resolves to.
|
|
498
|
+
|
|
499
|
+
https://docs.apify.com/api/v2#/reference/datasets/item-collection/get-items
|
|
500
|
+
|
|
501
|
+
Args:
|
|
502
|
+
offset: Number of items that should be skipped at the start. The default value is 0.
|
|
503
|
+
limit: Maximum number of dataset rows to scan. Fewer items are yielded when filters drop some, more
|
|
504
|
+
when `unwind` splits a row into several. By default there is no limit.
|
|
505
|
+
clean: If True, returns only non-empty items and skips hidden fields (i.e. fields starting with
|
|
506
|
+
the # character). The clean parameter is just a shortcut for skip_hidden=True and skip_empty=True
|
|
507
|
+
parameters.
|
|
508
|
+
fields: A list of fields which should be picked from the items, only these fields will remain in
|
|
509
|
+
the resulting record objects.
|
|
510
|
+
omit: A list of fields which should be omitted from the items.
|
|
511
|
+
unwind: A list of fields which should be unwound, in order which they should be processed. Each field
|
|
512
|
+
should be either an array or an object. If the field is an array then every element of the array
|
|
513
|
+
will become a separate record and merged with parent object. If the unwound field is an object then
|
|
514
|
+
it is merged with the parent object.
|
|
515
|
+
skip_empty: If True, then empty items are skipped from the output.
|
|
516
|
+
skip_hidden: If True, then hidden fields are skipped from the output, i.e. fields starting with
|
|
517
|
+
the # character.
|
|
518
|
+
chunk_size: Maximum number of dataset rows requested per API call.
|
|
519
|
+
poll_interval: How long to wait for the run to finish between polls.
|
|
520
|
+
timeout: Timeout for each API HTTP request.
|
|
521
|
+
|
|
522
|
+
Yields:
|
|
523
|
+
An item from the dataset.
|
|
524
|
+
"""
|
|
525
|
+
page_size = chunk_size or DEFAULT_CHUNK_SIZE
|
|
526
|
+
position = offset or 0
|
|
527
|
+
end = position + limit if limit else None
|
|
528
|
+
|
|
529
|
+
run = self.get(timeout=timeout)
|
|
530
|
+
# A `last_run()` client resolves `runs/last` per request, so a newer run would swap the dataset mid-iteration.
|
|
531
|
+
run_client = (
|
|
532
|
+
self._client_registry.run_client(
|
|
533
|
+
resource_id=run.id,
|
|
534
|
+
base_url=self._api_base_url,
|
|
535
|
+
public_base_url=self._public_base_url,
|
|
536
|
+
http_client=self._http_client,
|
|
537
|
+
client_registry=self._client_registry,
|
|
538
|
+
)
|
|
539
|
+
if run is not None and run.id != self._resource_id
|
|
540
|
+
else self
|
|
541
|
+
)
|
|
542
|
+
dataset_client = run_client.dataset()
|
|
543
|
+
|
|
544
|
+
def list_page(page_offset: int, page_limit: int) -> DatasetItemsPage:
|
|
545
|
+
return dataset_client.list_items(
|
|
546
|
+
offset=page_offset,
|
|
547
|
+
limit=page_limit,
|
|
548
|
+
clean=clean,
|
|
549
|
+
fields=fields,
|
|
550
|
+
omit=omit,
|
|
551
|
+
unwind=unwind,
|
|
552
|
+
skip_empty=skip_empty,
|
|
553
|
+
skip_hidden=skip_hidden,
|
|
554
|
+
timeout=timeout,
|
|
555
|
+
)
|
|
556
|
+
|
|
557
|
+
while True:
|
|
558
|
+
is_finished = run is None or run.status in _TERMINAL_STATUSES
|
|
559
|
+
dataset = dataset_client.get(timeout=timeout)
|
|
560
|
+
item_count = dataset.item_count if dataset else 0
|
|
561
|
+
if end is not None:
|
|
562
|
+
item_count = min(item_count, end)
|
|
563
|
+
|
|
564
|
+
while position < item_count:
|
|
565
|
+
page_limit = min(page_size, item_count - position)
|
|
566
|
+
page = list_page(position, page_limit)
|
|
567
|
+
yield from page.items
|
|
568
|
+
position += page_limit
|
|
569
|
+
|
|
570
|
+
if end is not None and position >= end:
|
|
571
|
+
return
|
|
572
|
+
if is_finished:
|
|
573
|
+
break
|
|
574
|
+
run = run_client.wait_for_finish(wait_duration=poll_interval, timeout=timeout)
|
|
575
|
+
|
|
576
|
+
while True:
|
|
577
|
+
page_limit = min(page_size, end - position) if end is not None else page_size
|
|
578
|
+
page = list_page(position, page_limit)
|
|
579
|
+
yield from page.items
|
|
580
|
+
# Only an empty page marks the end, as filters can shorten a full one. A page that `clean`, `skip_empty` or
|
|
581
|
+
# `unwind` emptied past a lagging `item_count` reports no scanned rows either, so a plain read checks.
|
|
582
|
+
if not page.count and (
|
|
583
|
+
not (clean or skip_empty or unwind)
|
|
584
|
+
or not dataset_client.list_items(offset=position, limit=1, timeout=timeout).items
|
|
585
|
+
):
|
|
586
|
+
return
|
|
587
|
+
position += page_limit
|
|
588
|
+
if end is not None and position >= end:
|
|
589
|
+
return
|
|
590
|
+
|
|
472
591
|
|
|
473
592
|
@docs_group('Resource clients')
|
|
474
593
|
class RunClientAsync(ResourceClientAsync):
|
|
@@ -903,3 +1022,121 @@ class RunClientAsync(ResourceClientAsync):
|
|
|
903
1022
|
to_logger = create_redirect_logger(f'apify.{name}')
|
|
904
1023
|
|
|
905
1024
|
return StatusMessageWatcherAsync(run_client=self, to_logger=to_logger, check_period=check_period)
|
|
1025
|
+
|
|
1026
|
+
async def iterate_dataset_items(
|
|
1027
|
+
self,
|
|
1028
|
+
*,
|
|
1029
|
+
offset: int | None = None,
|
|
1030
|
+
limit: int | None = None,
|
|
1031
|
+
clean: bool | None = None,
|
|
1032
|
+
fields: list[str] | None = None,
|
|
1033
|
+
omit: list[str] | None = None,
|
|
1034
|
+
unwind: list[str] | None = None,
|
|
1035
|
+
skip_empty: bool | None = None,
|
|
1036
|
+
skip_hidden: bool | None = None,
|
|
1037
|
+
chunk_size: int | None = None,
|
|
1038
|
+
poll_interval: timedelta = timedelta(seconds=5),
|
|
1039
|
+
timeout: Timeout = 'long',
|
|
1040
|
+
) -> AsyncIterator[dict]:
|
|
1041
|
+
"""Iterate over the items of the run's default dataset while the run is still producing them.
|
|
1042
|
+
|
|
1043
|
+
While the run has not finished, each poll yields the rows below the dataset's `item_count` and then waits up to
|
|
1044
|
+
`poll_interval` for the run to finish, so the last rows are read as soon as it does. Each page is requested with
|
|
1045
|
+
a `limit` that ends at `item_count`, so it covers exactly the rows it asks for, whatever the filters or `unwind`
|
|
1046
|
+
do to the items. `item_count` lags a few seconds behind the pushed items, so once the run reaches a terminal
|
|
1047
|
+
status, the rows past it are read a page at a time until none are left, and the iterator returns. On a
|
|
1048
|
+
`last_run()` client, the iterator sticks to the run that its first request resolves to.
|
|
1049
|
+
|
|
1050
|
+
https://docs.apify.com/api/v2#/reference/datasets/item-collection/get-items
|
|
1051
|
+
|
|
1052
|
+
Args:
|
|
1053
|
+
offset: Number of items that should be skipped at the start. The default value is 0.
|
|
1054
|
+
limit: Maximum number of dataset rows to scan. Fewer items are yielded when filters drop some, more
|
|
1055
|
+
when `unwind` splits a row into several. By default there is no limit.
|
|
1056
|
+
clean: If True, returns only non-empty items and skips hidden fields (i.e. fields starting with
|
|
1057
|
+
the # character). The clean parameter is just a shortcut for skip_hidden=True and skip_empty=True
|
|
1058
|
+
parameters.
|
|
1059
|
+
fields: A list of fields which should be picked from the items, only these fields will remain in
|
|
1060
|
+
the resulting record objects.
|
|
1061
|
+
omit: A list of fields which should be omitted from the items.
|
|
1062
|
+
unwind: A list of fields which should be unwound, in order which they should be processed. Each field
|
|
1063
|
+
should be either an array or an object. If the field is an array then every element of the array
|
|
1064
|
+
will become a separate record and merged with parent object. If the unwound field is an object then
|
|
1065
|
+
it is merged with the parent object.
|
|
1066
|
+
skip_empty: If True, then empty items are skipped from the output.
|
|
1067
|
+
skip_hidden: If True, then hidden fields are skipped from the output, i.e. fields starting with
|
|
1068
|
+
the # character.
|
|
1069
|
+
chunk_size: Maximum number of dataset rows requested per API call.
|
|
1070
|
+
poll_interval: How long to wait for the run to finish between polls.
|
|
1071
|
+
timeout: Timeout for each API HTTP request.
|
|
1072
|
+
|
|
1073
|
+
Yields:
|
|
1074
|
+
An item from the dataset.
|
|
1075
|
+
"""
|
|
1076
|
+
page_size = chunk_size or DEFAULT_CHUNK_SIZE
|
|
1077
|
+
position = offset or 0
|
|
1078
|
+
end = position + limit if limit else None
|
|
1079
|
+
|
|
1080
|
+
run = await self.get(timeout=timeout)
|
|
1081
|
+
# A `last_run()` client resolves `runs/last` per request, so a newer run would swap the dataset mid-iteration.
|
|
1082
|
+
run_client = (
|
|
1083
|
+
self._client_registry.run_client(
|
|
1084
|
+
resource_id=run.id,
|
|
1085
|
+
base_url=self._api_base_url,
|
|
1086
|
+
public_base_url=self._public_base_url,
|
|
1087
|
+
http_client=self._http_client,
|
|
1088
|
+
client_registry=self._client_registry,
|
|
1089
|
+
)
|
|
1090
|
+
if run is not None and run.id != self._resource_id
|
|
1091
|
+
else self
|
|
1092
|
+
)
|
|
1093
|
+
dataset_client = run_client.dataset()
|
|
1094
|
+
|
|
1095
|
+
async def list_page(page_offset: int, page_limit: int) -> DatasetItemsPage:
|
|
1096
|
+
return await dataset_client.list_items(
|
|
1097
|
+
offset=page_offset,
|
|
1098
|
+
limit=page_limit,
|
|
1099
|
+
clean=clean,
|
|
1100
|
+
fields=fields,
|
|
1101
|
+
omit=omit,
|
|
1102
|
+
unwind=unwind,
|
|
1103
|
+
skip_empty=skip_empty,
|
|
1104
|
+
skip_hidden=skip_hidden,
|
|
1105
|
+
timeout=timeout,
|
|
1106
|
+
)
|
|
1107
|
+
|
|
1108
|
+
while True:
|
|
1109
|
+
is_finished = run is None or run.status in _TERMINAL_STATUSES
|
|
1110
|
+
dataset = await dataset_client.get(timeout=timeout)
|
|
1111
|
+
item_count = dataset.item_count if dataset else 0
|
|
1112
|
+
if end is not None:
|
|
1113
|
+
item_count = min(item_count, end)
|
|
1114
|
+
|
|
1115
|
+
while position < item_count:
|
|
1116
|
+
page_limit = min(page_size, item_count - position)
|
|
1117
|
+
page = await list_page(position, page_limit)
|
|
1118
|
+
for item in page.items:
|
|
1119
|
+
yield item
|
|
1120
|
+
position += page_limit
|
|
1121
|
+
|
|
1122
|
+
if end is not None and position >= end:
|
|
1123
|
+
return
|
|
1124
|
+
if is_finished:
|
|
1125
|
+
break
|
|
1126
|
+
run = await run_client.wait_for_finish(wait_duration=poll_interval, timeout=timeout)
|
|
1127
|
+
|
|
1128
|
+
while True:
|
|
1129
|
+
page_limit = min(page_size, end - position) if end is not None else page_size
|
|
1130
|
+
page = await list_page(position, page_limit)
|
|
1131
|
+
for item in page.items:
|
|
1132
|
+
yield item
|
|
1133
|
+
# Only an empty page marks the end, as filters can shorten a full one. A page that `clean`, `skip_empty` or
|
|
1134
|
+
# `unwind` emptied past a lagging `item_count` reports no scanned rows either, so a plain read checks.
|
|
1135
|
+
if not page.count and (
|
|
1136
|
+
not (clean or skip_empty or unwind)
|
|
1137
|
+
or not (await dataset_client.list_items(offset=position, limit=1, timeout=timeout)).items
|
|
1138
|
+
):
|
|
1139
|
+
return
|
|
1140
|
+
position += page_limit
|
|
1141
|
+
if end is not None and position >= end:
|
|
1142
|
+
return
|
|
@@ -18,6 +18,7 @@ from apify_client._resource_clients._resource_client import ResourceClient, Reso
|
|
|
18
18
|
from apify_client._utils.encoding import encode_webhooks_to_base64
|
|
19
19
|
from apify_client._utils.http import response_to_dict
|
|
20
20
|
from apify_client._utils.time import to_seconds
|
|
21
|
+
from apify_client._utils.wait_for_resources import start_waiting_for_resources, start_waiting_for_resources_async
|
|
21
22
|
|
|
22
23
|
if TYPE_CHECKING:
|
|
23
24
|
from datetime import timedelta
|
|
@@ -221,6 +222,7 @@ class TaskClient(ResourceClient):
|
|
|
221
222
|
restart_on_error: bool | None = None,
|
|
222
223
|
wait_for_finish: int | None = None,
|
|
223
224
|
webhooks: WebhooksList | None = None,
|
|
225
|
+
wait_for_resources: bool | timedelta = False,
|
|
224
226
|
timeout: Timeout = 'medium',
|
|
225
227
|
) -> Run:
|
|
226
228
|
"""Start the task and immediately return the Run object.
|
|
@@ -248,6 +250,13 @@ class TaskClient(ResourceClient):
|
|
|
248
250
|
* `event_types`: List of `WebhookEventType` values which trigger the webhook.
|
|
249
251
|
* `request_url`: URL to which to send the webhook HTTP request.
|
|
250
252
|
* `payload_template`: Optional template for the request payload.
|
|
253
|
+
wait_for_resources: Retry the start while the account lacks the memory or a concurrent-run slot for the run,
|
|
254
|
+
that is while the API rejects it with an `ApifyApiError` of type `actor-memory-limit-exceeded` or
|
|
255
|
+
`concurrent-runs-limit-exceeded`. Both clear as other runs or builds finish. The start is retried
|
|
256
|
+
every 10 seconds, and any other error is raised right away. `True` retries until the run starts, a
|
|
257
|
+
`timedelta` stops retrying after that long and raises the last error. A run that requests more memory
|
|
258
|
+
than the whole memory limit of the account is rejected with `actor-memory-limit-exceeded` as well and
|
|
259
|
+
never starts, so `True` retries it forever.
|
|
251
260
|
timeout: Timeout for the API HTTP request.
|
|
252
261
|
|
|
253
262
|
Returns:
|
|
@@ -266,13 +275,16 @@ class TaskClient(ResourceClient):
|
|
|
266
275
|
webhooks=encode_webhooks_to_base64(webhooks),
|
|
267
276
|
)
|
|
268
277
|
|
|
269
|
-
response =
|
|
270
|
-
|
|
271
|
-
|
|
272
|
-
|
|
273
|
-
|
|
274
|
-
|
|
275
|
-
|
|
278
|
+
response = start_waiting_for_resources(
|
|
279
|
+
lambda: self._http_client.call(
|
|
280
|
+
url=self._build_url('runs'),
|
|
281
|
+
method='POST',
|
|
282
|
+
headers={'content-type': 'application/json; charset=utf-8'},
|
|
283
|
+
json=task_input.model_dump() if task_input is not None else None,
|
|
284
|
+
params=request_params,
|
|
285
|
+
timeout=timeout,
|
|
286
|
+
),
|
|
287
|
+
wait_for_resources=wait_for_resources,
|
|
276
288
|
)
|
|
277
289
|
|
|
278
290
|
result = response_to_dict(response)
|
|
@@ -289,6 +301,7 @@ class TaskClient(ResourceClient):
|
|
|
289
301
|
restart_on_error: bool | None = None,
|
|
290
302
|
webhooks: WebhooksList | None = None,
|
|
291
303
|
wait_duration: timedelta | None = None,
|
|
304
|
+
wait_for_resources: bool | timedelta = False,
|
|
292
305
|
timeout: Timeout = 'no_timeout',
|
|
293
306
|
) -> Run | None:
|
|
294
307
|
"""Start a task and wait for it to finish before returning the Run object.
|
|
@@ -314,6 +327,14 @@ class TaskClient(ResourceClient):
|
|
|
314
327
|
the Actor or task, you do not have to add it again here.
|
|
315
328
|
wait_duration: The maximum time the server waits for the task run to finish. If not provided,
|
|
316
329
|
waits indefinitely.
|
|
330
|
+
wait_for_resources: Retry the start while the account lacks the memory or a concurrent-run slot for the run,
|
|
331
|
+
that is while the API rejects it with an `ApifyApiError` of type `actor-memory-limit-exceeded` or
|
|
332
|
+
`concurrent-runs-limit-exceeded`. Both clear as other runs or builds finish. The start is retried
|
|
333
|
+
every 10 seconds, and any other error is raised right away. `True` retries until the run starts, a
|
|
334
|
+
`timedelta` stops retrying after that long and raises the last error. A run that requests more memory
|
|
335
|
+
than the whole memory limit of the account is rejected with `actor-memory-limit-exceeded` as well and
|
|
336
|
+
never starts, so `True` retries it forever. The time spent retrying doesn't count toward
|
|
337
|
+
`wait_duration`.
|
|
317
338
|
timeout: Timeout for the API HTTP request.
|
|
318
339
|
|
|
319
340
|
Returns:
|
|
@@ -327,6 +348,7 @@ class TaskClient(ResourceClient):
|
|
|
327
348
|
run_timeout=run_timeout,
|
|
328
349
|
restart_on_error=restart_on_error,
|
|
329
350
|
webhooks=webhooks,
|
|
351
|
+
wait_for_resources=wait_for_resources,
|
|
330
352
|
timeout=timeout,
|
|
331
353
|
)
|
|
332
354
|
|
|
@@ -602,6 +624,7 @@ class TaskClientAsync(ResourceClientAsync):
|
|
|
602
624
|
restart_on_error: bool | None = None,
|
|
603
625
|
wait_for_finish: int | None = None,
|
|
604
626
|
webhooks: WebhooksList | None = None,
|
|
627
|
+
wait_for_resources: bool | timedelta = False,
|
|
605
628
|
timeout: Timeout = 'medium',
|
|
606
629
|
) -> Run:
|
|
607
630
|
"""Start the task and immediately return the Run object.
|
|
@@ -629,6 +652,13 @@ class TaskClientAsync(ResourceClientAsync):
|
|
|
629
652
|
* `event_types`: List of `WebhookEventType` values which trigger the webhook.
|
|
630
653
|
* `request_url`: URL to which to send the webhook HTTP request.
|
|
631
654
|
* `payload_template`: Optional template for the request payload.
|
|
655
|
+
wait_for_resources: Retry the start while the account lacks the memory or a concurrent-run slot for the run,
|
|
656
|
+
that is while the API rejects it with an `ApifyApiError` of type `actor-memory-limit-exceeded` or
|
|
657
|
+
`concurrent-runs-limit-exceeded`. Both clear as other runs or builds finish. The start is retried
|
|
658
|
+
every 10 seconds, and any other error is raised right away. `True` retries until the run starts, a
|
|
659
|
+
`timedelta` stops retrying after that long and raises the last error. A run that requests more memory
|
|
660
|
+
than the whole memory limit of the account is rejected with `actor-memory-limit-exceeded` as well and
|
|
661
|
+
never starts, so `True` retries it forever.
|
|
632
662
|
timeout: Timeout for the API HTTP request.
|
|
633
663
|
|
|
634
664
|
Returns:
|
|
@@ -647,13 +677,16 @@ class TaskClientAsync(ResourceClientAsync):
|
|
|
647
677
|
webhooks=encode_webhooks_to_base64(webhooks),
|
|
648
678
|
)
|
|
649
679
|
|
|
650
|
-
response = await
|
|
651
|
-
|
|
652
|
-
|
|
653
|
-
|
|
654
|
-
|
|
655
|
-
|
|
656
|
-
|
|
680
|
+
response = await start_waiting_for_resources_async(
|
|
681
|
+
lambda: self._http_client.call(
|
|
682
|
+
url=self._build_url('runs'),
|
|
683
|
+
method='POST',
|
|
684
|
+
headers={'content-type': 'application/json; charset=utf-8'},
|
|
685
|
+
json=task_input.model_dump() if task_input is not None else None,
|
|
686
|
+
params=request_params,
|
|
687
|
+
timeout=timeout,
|
|
688
|
+
),
|
|
689
|
+
wait_for_resources=wait_for_resources,
|
|
657
690
|
)
|
|
658
691
|
|
|
659
692
|
result = response_to_dict(response)
|
|
@@ -670,6 +703,7 @@ class TaskClientAsync(ResourceClientAsync):
|
|
|
670
703
|
restart_on_error: bool | None = None,
|
|
671
704
|
webhooks: WebhooksList | None = None,
|
|
672
705
|
wait_duration: timedelta | None = None,
|
|
706
|
+
wait_for_resources: bool | timedelta = False,
|
|
673
707
|
timeout: Timeout = 'no_timeout',
|
|
674
708
|
) -> Run | None:
|
|
675
709
|
"""Start a task and wait for it to finish before returning the Run object.
|
|
@@ -695,6 +729,14 @@ class TaskClientAsync(ResourceClientAsync):
|
|
|
695
729
|
the Actor or task, you do not have to add it again here.
|
|
696
730
|
wait_duration: The maximum time the server waits for the task run to finish. If not provided,
|
|
697
731
|
waits indefinitely.
|
|
732
|
+
wait_for_resources: Retry the start while the account lacks the memory or a concurrent-run slot for the run,
|
|
733
|
+
that is while the API rejects it with an `ApifyApiError` of type `actor-memory-limit-exceeded` or
|
|
734
|
+
`concurrent-runs-limit-exceeded`. Both clear as other runs or builds finish. The start is retried
|
|
735
|
+
every 10 seconds, and any other error is raised right away. `True` retries until the run starts, a
|
|
736
|
+
`timedelta` stops retrying after that long and raises the last error. A run that requests more memory
|
|
737
|
+
than the whole memory limit of the account is rejected with `actor-memory-limit-exceeded` as well and
|
|
738
|
+
never starts, so `True` retries it forever. The time spent retrying doesn't count toward
|
|
739
|
+
`wait_duration`.
|
|
698
740
|
timeout: Timeout for the API HTTP request.
|
|
699
741
|
|
|
700
742
|
Returns:
|
|
@@ -708,6 +750,7 @@ class TaskClientAsync(ResourceClientAsync):
|
|
|
708
750
|
run_timeout=run_timeout,
|
|
709
751
|
restart_on_error=restart_on_error,
|
|
710
752
|
webhooks=webhooks,
|
|
753
|
+
wait_for_resources=wait_for_resources,
|
|
711
754
|
timeout=timeout,
|
|
712
755
|
)
|
|
713
756
|
run_client = self._client_registry.run_client(
|
|
@@ -0,0 +1,82 @@
|
|
|
1
|
+
from __future__ import annotations
|
|
2
|
+
|
|
3
|
+
import asyncio
|
|
4
|
+
import time
|
|
5
|
+
from datetime import timedelta
|
|
6
|
+
from typing import TYPE_CHECKING, Any, TypeVar
|
|
7
|
+
|
|
8
|
+
from apify_client._logging import logger
|
|
9
|
+
from apify_client.errors import ApifyApiError
|
|
10
|
+
from apify_client.http_clients._streamed_body import StreamedRequestBody
|
|
11
|
+
|
|
12
|
+
if TYPE_CHECKING:
|
|
13
|
+
from collections.abc import Awaitable, Callable
|
|
14
|
+
|
|
15
|
+
T = TypeVar('T')
|
|
16
|
+
|
|
17
|
+
RESOURCE_LIMIT_ERROR_TYPES = frozenset({'actor-memory-limit-exceeded', 'concurrent-runs-limit-exceeded'})
|
|
18
|
+
"""Error types the API rejects a run start with while the account has no free memory or concurrent-run slot for it.
|
|
19
|
+
|
|
20
|
+
Both clear as other runs or builds finish.
|
|
21
|
+
"""
|
|
22
|
+
|
|
23
|
+
WAIT_FOR_RESOURCES_COOLDOWN = timedelta(seconds=10)
|
|
24
|
+
"""Cooldown between two attempts to start a run that was rejected for lack of resources."""
|
|
25
|
+
|
|
26
|
+
|
|
27
|
+
def prepare_resendable_body(data: Any, *, wait_for_resources: bool | timedelta) -> tuple[Any, bool | timedelta]:
|
|
28
|
+
"""Wrap a streamed request body once, so every start attempt sends the same body.
|
|
29
|
+
|
|
30
|
+
A seekable `io.IOBase` source is rewound before each attempt. Any other streamed source is used up by the first
|
|
31
|
+
attempt, so the returned `wait_for_resources` is `False` and a start rejected for lack of resources is not retried.
|
|
32
|
+
"""
|
|
33
|
+
if wait_for_resources is False or not StreamedRequestBody.is_streamable(data):
|
|
34
|
+
return data, wait_for_resources
|
|
35
|
+
body = data if isinstance(data, StreamedRequestBody) else StreamedRequestBody(data)
|
|
36
|
+
return body, wait_for_resources if body.rewindable else False
|
|
37
|
+
|
|
38
|
+
|
|
39
|
+
def _next_delay(exc: ApifyApiError, deadline: float | None) -> float:
|
|
40
|
+
"""Return the seconds to sleep before the next attempt, or re-raise `exc` if no attempt should follow."""
|
|
41
|
+
if exc.type not in RESOURCE_LIMIT_ERROR_TYPES:
|
|
42
|
+
raise exc
|
|
43
|
+
delay = WAIT_FOR_RESOURCES_COOLDOWN.total_seconds()
|
|
44
|
+
if deadline is not None:
|
|
45
|
+
remaining = deadline - time.monotonic()
|
|
46
|
+
if remaining <= 0:
|
|
47
|
+
raise exc
|
|
48
|
+
delay = min(delay, remaining)
|
|
49
|
+
logger.info('Not enough resources to start the run, retrying in %.3gs: %s', delay, exc.message)
|
|
50
|
+
return delay
|
|
51
|
+
|
|
52
|
+
|
|
53
|
+
def start_waiting_for_resources(start: Callable[[], T], *, wait_for_resources: bool | timedelta) -> T:
|
|
54
|
+
"""Make the `start` request, retrying it every `WAIT_FOR_RESOURCES_COOLDOWN` while it fails for lack of resources.
|
|
55
|
+
|
|
56
|
+
`True` retries until the request succeeds, a `timedelta` bounds the retrying, after which the last error is raised.
|
|
57
|
+
Any other error is raised right away.
|
|
58
|
+
"""
|
|
59
|
+
if wait_for_resources is False:
|
|
60
|
+
return start()
|
|
61
|
+
deadline = None if wait_for_resources is True else time.monotonic() + wait_for_resources.total_seconds()
|
|
62
|
+
while True:
|
|
63
|
+
try:
|
|
64
|
+
return start()
|
|
65
|
+
except ApifyApiError as exc:
|
|
66
|
+
time.sleep(_next_delay(exc, deadline))
|
|
67
|
+
|
|
68
|
+
|
|
69
|
+
async def start_waiting_for_resources_async(
|
|
70
|
+
start: Callable[[], Awaitable[T]],
|
|
71
|
+
*,
|
|
72
|
+
wait_for_resources: bool | timedelta,
|
|
73
|
+
) -> T:
|
|
74
|
+
"""Async variant of `start_waiting_for_resources`."""
|
|
75
|
+
if wait_for_resources is False:
|
|
76
|
+
return await start()
|
|
77
|
+
deadline = None if wait_for_resources is True else time.monotonic() + wait_for_resources.total_seconds()
|
|
78
|
+
while True:
|
|
79
|
+
try:
|
|
80
|
+
return await start()
|
|
81
|
+
except ApifyApiError as exc:
|
|
82
|
+
await asyncio.sleep(_next_delay(exc, deadline))
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
{apify_client-3.2.2b1 → apify_client-3.2.2b3}/src/apify_client/_resource_clients/__init__.py
RENAMED
|
File without changes
|
{apify_client-3.2.2b1 → apify_client-3.2.2b3}/src/apify_client/_resource_clients/actor_collection.py
RENAMED
|
File without changes
|
{apify_client-3.2.2b1 → apify_client-3.2.2b3}/src/apify_client/_resource_clients/actor_env_var.py
RENAMED
|
File without changes
|
|
File without changes
|
{apify_client-3.2.2b1 → apify_client-3.2.2b3}/src/apify_client/_resource_clients/actor_version.py
RENAMED
|
File without changes
|
|
File without changes
|
|
File without changes
|
{apify_client-3.2.2b1 → apify_client-3.2.2b3}/src/apify_client/_resource_clients/build_collection.py
RENAMED
|
File without changes
|
|
File without changes
|
|
File without changes
|
{apify_client-3.2.2b1 → apify_client-3.2.2b3}/src/apify_client/_resource_clients/key_value_store.py
RENAMED
|
File without changes
|
|
File without changes
|
|
File without changes
|
{apify_client-3.2.2b1 → apify_client-3.2.2b3}/src/apify_client/_resource_clients/request_queue.py
RENAMED
|
File without changes
|
|
File without changes
|
{apify_client-3.2.2b1 → apify_client-3.2.2b3}/src/apify_client/_resource_clients/run_collection.py
RENAMED
|
File without changes
|
{apify_client-3.2.2b1 → apify_client-3.2.2b3}/src/apify_client/_resource_clients/schedule.py
RENAMED
|
File without changes
|
|
File without changes
|
{apify_client-3.2.2b1 → apify_client-3.2.2b3}/src/apify_client/_resource_clients/store_collection.py
RENAMED
|
File without changes
|
{apify_client-3.2.2b1 → apify_client-3.2.2b3}/src/apify_client/_resource_clients/task_collection.py
RENAMED
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
{apify_client-3.2.2b1 → apify_client-3.2.2b3}/src/apify_client/_resource_clients/webhook_dispatch.py
RENAMED
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
{apify_client-3.2.2b1 → apify_client-3.2.2b3}/src/apify_client/http_clients/_streamed_body.py
RENAMED
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|