apify-client 3.2.2b2__tar.gz → 3.2.2b3__tar.gz

This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
Files changed (70) hide show
  1. {apify_client-3.2.2b2 → apify_client-3.2.2b3}/CHANGELOG.md +1 -0
  2. {apify_client-3.2.2b2 → apify_client-3.2.2b3}/PKG-INFO +1 -1
  3. {apify_client-3.2.2b2 → apify_client-3.2.2b3}/pyproject.toml +1 -1
  4. {apify_client-3.2.2b2 → apify_client-3.2.2b3}/pyproject.toml.orig +1 -1
  5. {apify_client-3.2.2b2 → apify_client-3.2.2b3}/src/apify_client/_resource_clients/_resource_client.py +11 -1
  6. {apify_client-3.2.2b2 → apify_client-3.2.2b3}/src/apify_client/_resource_clients/run.py +238 -1
  7. {apify_client-3.2.2b2 → apify_client-3.2.2b3}/CONTRIBUTING.md +0 -0
  8. {apify_client-3.2.2b2 → apify_client-3.2.2b3}/LICENSE +0 -0
  9. {apify_client-3.2.2b2 → apify_client-3.2.2b3}/README.md +0 -0
  10. {apify_client-3.2.2b2 → apify_client-3.2.2b3}/src/apify_client/__init__.py +0 -0
  11. {apify_client-3.2.2b2 → apify_client-3.2.2b3}/src/apify_client/_apify_client.py +0 -0
  12. {apify_client-3.2.2b2 → apify_client-3.2.2b3}/src/apify_client/_client_registry.py +0 -0
  13. {apify_client-3.2.2b2 → apify_client-3.2.2b3}/src/apify_client/_consts.py +0 -0
  14. {apify_client-3.2.2b2 → apify_client-3.2.2b3}/src/apify_client/_docs.py +0 -0
  15. {apify_client-3.2.2b2 → apify_client-3.2.2b3}/src/apify_client/_literals.py +0 -0
  16. {apify_client-3.2.2b2 → apify_client-3.2.2b3}/src/apify_client/_logging.py +0 -0
  17. {apify_client-3.2.2b2 → apify_client-3.2.2b3}/src/apify_client/_models.py +0 -0
  18. {apify_client-3.2.2b2 → apify_client-3.2.2b3}/src/apify_client/_pagination.py +0 -0
  19. {apify_client-3.2.2b2 → apify_client-3.2.2b3}/src/apify_client/_resource_clients/__init__.py +0 -0
  20. {apify_client-3.2.2b2 → apify_client-3.2.2b3}/src/apify_client/_resource_clients/actor.py +0 -0
  21. {apify_client-3.2.2b2 → apify_client-3.2.2b3}/src/apify_client/_resource_clients/actor_collection.py +0 -0
  22. {apify_client-3.2.2b2 → apify_client-3.2.2b3}/src/apify_client/_resource_clients/actor_env_var.py +0 -0
  23. {apify_client-3.2.2b2 → apify_client-3.2.2b3}/src/apify_client/_resource_clients/actor_env_var_collection.py +0 -0
  24. {apify_client-3.2.2b2 → apify_client-3.2.2b3}/src/apify_client/_resource_clients/actor_version.py +0 -0
  25. {apify_client-3.2.2b2 → apify_client-3.2.2b3}/src/apify_client/_resource_clients/actor_version_collection.py +0 -0
  26. {apify_client-3.2.2b2 → apify_client-3.2.2b3}/src/apify_client/_resource_clients/build.py +0 -0
  27. {apify_client-3.2.2b2 → apify_client-3.2.2b3}/src/apify_client/_resource_clients/build_collection.py +0 -0
  28. {apify_client-3.2.2b2 → apify_client-3.2.2b3}/src/apify_client/_resource_clients/dataset.py +0 -0
  29. {apify_client-3.2.2b2 → apify_client-3.2.2b3}/src/apify_client/_resource_clients/dataset_collection.py +0 -0
  30. {apify_client-3.2.2b2 → apify_client-3.2.2b3}/src/apify_client/_resource_clients/key_value_store.py +0 -0
  31. {apify_client-3.2.2b2 → apify_client-3.2.2b3}/src/apify_client/_resource_clients/key_value_store_collection.py +0 -0
  32. {apify_client-3.2.2b2 → apify_client-3.2.2b3}/src/apify_client/_resource_clients/log.py +0 -0
  33. {apify_client-3.2.2b2 → apify_client-3.2.2b3}/src/apify_client/_resource_clients/request_queue.py +0 -0
  34. {apify_client-3.2.2b2 → apify_client-3.2.2b3}/src/apify_client/_resource_clients/request_queue_collection.py +0 -0
  35. {apify_client-3.2.2b2 → apify_client-3.2.2b3}/src/apify_client/_resource_clients/run_collection.py +0 -0
  36. {apify_client-3.2.2b2 → apify_client-3.2.2b3}/src/apify_client/_resource_clients/schedule.py +0 -0
  37. {apify_client-3.2.2b2 → apify_client-3.2.2b3}/src/apify_client/_resource_clients/schedule_collection.py +0 -0
  38. {apify_client-3.2.2b2 → apify_client-3.2.2b3}/src/apify_client/_resource_clients/store_collection.py +0 -0
  39. {apify_client-3.2.2b2 → apify_client-3.2.2b3}/src/apify_client/_resource_clients/task.py +0 -0
  40. {apify_client-3.2.2b2 → apify_client-3.2.2b3}/src/apify_client/_resource_clients/task_collection.py +0 -0
  41. {apify_client-3.2.2b2 → apify_client-3.2.2b3}/src/apify_client/_resource_clients/user.py +0 -0
  42. {apify_client-3.2.2b2 → apify_client-3.2.2b3}/src/apify_client/_resource_clients/webhook.py +0 -0
  43. {apify_client-3.2.2b2 → apify_client-3.2.2b3}/src/apify_client/_resource_clients/webhook_collection.py +0 -0
  44. {apify_client-3.2.2b2 → apify_client-3.2.2b3}/src/apify_client/_resource_clients/webhook_dispatch.py +0 -0
  45. {apify_client-3.2.2b2 → apify_client-3.2.2b3}/src/apify_client/_resource_clients/webhook_dispatch_collection.py +0 -0
  46. {apify_client-3.2.2b2 → apify_client-3.2.2b3}/src/apify_client/_statistics.py +0 -0
  47. {apify_client-3.2.2b2 → apify_client-3.2.2b3}/src/apify_client/_status_message_watcher.py +0 -0
  48. {apify_client-3.2.2b2 → apify_client-3.2.2b3}/src/apify_client/_streamed_log.py +0 -0
  49. {apify_client-3.2.2b2 → apify_client-3.2.2b3}/src/apify_client/_typeddicts.py +0 -0
  50. {apify_client-3.2.2b2 → apify_client-3.2.2b3}/src/apify_client/_utils/__init__.py +0 -0
  51. {apify_client-3.2.2b2 → apify_client-3.2.2b3}/src/apify_client/_utils/crypto.py +0 -0
  52. {apify_client-3.2.2b2 → apify_client-3.2.2b3}/src/apify_client/_utils/encoding.py +0 -0
  53. {apify_client-3.2.2b2 → apify_client-3.2.2b3}/src/apify_client/_utils/errors.py +0 -0
  54. {apify_client-3.2.2b2 → apify_client-3.2.2b3}/src/apify_client/_utils/http.py +0 -0
  55. {apify_client-3.2.2b2 → apify_client-3.2.2b3}/src/apify_client/_utils/time.py +0 -0
  56. {apify_client-3.2.2b2 → apify_client-3.2.2b3}/src/apify_client/_utils/try_import.py +0 -0
  57. {apify_client-3.2.2b2 → apify_client-3.2.2b3}/src/apify_client/_utils/wait_for_resources.py +0 -0
  58. {apify_client-3.2.2b2 → apify_client-3.2.2b3}/src/apify_client/errors.py +0 -0
  59. {apify_client-3.2.2b2 → apify_client-3.2.2b3}/src/apify_client/http_clients/__init__.py +0 -0
  60. {apify_client-3.2.2b2 → apify_client-3.2.2b3}/src/apify_client/http_clients/_base.py +0 -0
  61. {apify_client-3.2.2b2 → apify_client-3.2.2b3}/src/apify_client/http_clients/_httpx2.py +0 -0
  62. {apify_client-3.2.2b2 → apify_client-3.2.2b3}/src/apify_client/http_clients/_impit.py +0 -0
  63. {apify_client-3.2.2b2 → apify_client-3.2.2b3}/src/apify_client/http_clients/_streamed_body.py +0 -0
  64. {apify_client-3.2.2b2 → apify_client-3.2.2b3}/src/apify_client/http_compressors/__init__.py +0 -0
  65. {apify_client-3.2.2b2 → apify_client-3.2.2b3}/src/apify_client/http_compressors/_base.py +0 -0
  66. {apify_client-3.2.2b2 → apify_client-3.2.2b3}/src/apify_client/http_compressors/_brotli.py +0 -0
  67. {apify_client-3.2.2b2 → apify_client-3.2.2b3}/src/apify_client/http_compressors/_gzip.py +0 -0
  68. {apify_client-3.2.2b2 → apify_client-3.2.2b3}/src/apify_client/http_compressors/_resolve.py +0 -0
  69. {apify_client-3.2.2b2 → apify_client-3.2.2b3}/src/apify_client/py.typed +0 -0
  70. {apify_client-3.2.2b2 → apify_client-3.2.2b3}/src/apify_client/types.py +0 -0
@@ -9,6 +9,7 @@ All notable changes to this project will be documented in this file.
9
9
 
10
10
  - Stream request bodies from files, iterables, and responses ([#1060](https://github.com/apify/apify-client-python/pull/1060)) ([9be215b](https://github.com/apify/apify-client-python/commit/9be215b23a13de0abc4a255a18450cc764e7a6c7)) by [@vdusek](https://github.com/vdusek), closes [#972](https://github.com/apify/apify-client-python/issues/972)
11
11
  - Retry starting a run on memory and concurrent-runs limits ([#1081](https://github.com/apify/apify-client-python/pull/1081)) ([3dd8ca5](https://github.com/apify/apify-client-python/commit/3dd8ca5fe4d2537cb009cb310ba12b39927a9558)) by [@vdusek](https://github.com/vdusek), closes [#1071](https://github.com/apify/apify-client-python/issues/1071)
12
+ - Add live iteration over a run's dataset items ([#1079](https://github.com/apify/apify-client-python/pull/1079)) ([6aa8ead](https://github.com/apify/apify-client-python/commit/6aa8ead676a3520f6a418597eb62f319a3d19d30)) by [@vdusek](https://github.com/vdusek), closes [#1065](https://github.com/apify/apify-client-python/issues/1065)
12
13
 
13
14
  ### 🐛 Bug Fixes
14
15
 
@@ -1,6 +1,6 @@
1
1
  Metadata-Version: 2.4
2
2
  Name: apify_client
3
- Version: 3.2.2b2
3
+ Version: 3.2.2b3
4
4
  Summary: Apify API client for Python
5
5
  Keywords: apify,api,client,automation,crawling,scraping
6
6
  Author: Apify Technologies s.r.o.
@@ -4,7 +4,7 @@ build-backend = "uv_build"
4
4
 
5
5
  [project]
6
6
  name = "apify_client"
7
- version = "3.2.2b2"
7
+ version = "3.2.2b3"
8
8
  description = "Apify API client for Python"
9
9
  license = "Apache-2.0"
10
10
  license-files = ["LICENSE"]
@@ -4,7 +4,7 @@ build-backend = "uv_build"
4
4
 
5
5
  [project]
6
6
  name = "apify_client"
7
- version = "3.2.2b2"
7
+ version = "3.2.2b3"
8
8
  description = "Apify API client for Python"
9
9
  authors = [{ name = "Apify Technologies s.r.o.", email = "support@apify.com" }]
10
10
  license = "Apache-2.0"
@@ -42,6 +42,7 @@ class ResourceClientBase(metaclass=WithLogDetailsClient):
42
42
  client_registry: Any,
43
43
  resource_id: str | None = None,
44
44
  params: dict | None = None,
45
+ api_base_url: str | None = None,
45
46
  ) -> None:
46
47
  """Initialize the resource client.
47
48
 
@@ -53,11 +54,13 @@ class ResourceClientBase(metaclass=WithLogDetailsClient):
53
54
  client_registry: Bundle of client classes for dependency injection.
54
55
  resource_id: Optional resource ID for single-resource clients.
55
56
  params: Optional default parameters for all requests.
57
+ api_base_url: Base URL of the API itself, for clients of top-level resources. Defaults to `base_url`.
56
58
  """
57
59
  if resource_path.endswith('/'):
58
60
  raise ValueError('resource_path must not end with "/"')
59
61
 
60
62
  self._base_url = base_url
63
+ self._api_base_url = api_base_url or base_url
61
64
  self._public_base_url = public_base_url
62
65
  self._http_client = http_client
63
66
  self._default_params = params or {}
@@ -82,11 +85,12 @@ class ResourceClientBase(metaclass=WithLogDetailsClient):
82
85
  def _base_client_kwargs(self) -> dict[str, Any]:
83
86
  """Base kwargs for creating nested/child clients.
84
87
 
85
- Returns dict with base_url, public_base_url, http_client, and client_registry. Caller adds
88
+ Returns dict with base_url, api_base_url, public_base_url, http_client, and client_registry. Caller adds
86
89
  resource_path, resource_id, and params as needed.
87
90
  """
88
91
  return {
89
92
  'base_url': self._resource_url,
93
+ 'api_base_url': self._api_base_url,
90
94
  'public_base_url': self._public_base_url,
91
95
  'http_client': self._http_client,
92
96
  'client_registry': self._client_registry,
@@ -197,6 +201,7 @@ class ResourceClient(ResourceClientBase):
197
201
  client_registry: ClientRegistry,
198
202
  resource_id: str | None = None,
199
203
  params: dict | None = None,
204
+ api_base_url: str | None = None,
200
205
  ) -> None:
201
206
  """Initialize the resource client.
202
207
 
@@ -208,6 +213,7 @@ class ResourceClient(ResourceClientBase):
208
213
  client_registry: Bundle of client classes for dependency injection.
209
214
  resource_id: Optional resource ID for single-resource clients.
210
215
  params: Optional default parameters for all requests.
216
+ api_base_url: Base URL of the API itself, for clients of top-level resources. Defaults to `base_url`.
211
217
  """
212
218
  super().__init__(
213
219
  base_url=base_url,
@@ -217,6 +223,7 @@ class ResourceClient(ResourceClientBase):
217
223
  client_registry=client_registry,
218
224
  resource_id=resource_id,
219
225
  params=params,
226
+ api_base_url=api_base_url,
220
227
  )
221
228
 
222
229
  def _get(self, *, timeout: Timeout) -> dict | None:
@@ -389,6 +396,7 @@ class ResourceClientAsync(ResourceClientBase):
389
396
  client_registry: ClientRegistryAsync,
390
397
  resource_id: str | None = None,
391
398
  params: dict | None = None,
399
+ api_base_url: str | None = None,
392
400
  ) -> None:
393
401
  """Initialize the resource client.
394
402
 
@@ -400,6 +408,7 @@ class ResourceClientAsync(ResourceClientBase):
400
408
  client_registry: Bundle of client classes for dependency injection.
401
409
  resource_id: Optional resource ID for single-resource clients.
402
410
  params: Optional default parameters for all requests.
411
+ api_base_url: Base URL of the API itself, for clients of top-level resources. Defaults to `base_url`.
403
412
  """
404
413
  super().__init__(
405
414
  base_url=base_url,
@@ -409,6 +418,7 @@ class ResourceClientAsync(ResourceClientBase):
409
418
  client_registry=client_registry,
410
419
  resource_id=resource_id,
411
420
  params=params,
421
+ api_base_url=api_base_url,
412
422
  )
413
423
 
414
424
  async def _get(self, *, timeout: Timeout) -> dict | None:
@@ -10,7 +10,8 @@ from typing import TYPE_CHECKING, Any
10
10
  from apify_client._docs import docs_group
11
11
  from apify_client._logging import create_redirect_logger
12
12
  from apify_client._models import Run, RunResponse
13
- from apify_client._resource_clients._resource_client import ResourceClient, ResourceClientAsync
13
+ from apify_client._pagination import DEFAULT_CHUNK_SIZE
14
+ from apify_client._resource_clients._resource_client import _TERMINAL_STATUSES, ResourceClient, ResourceClientAsync
14
15
  from apify_client._status_message_watcher import StatusMessageWatcher, StatusMessageWatcherAsync
15
16
  from apify_client._streamed_log import StreamedLog, StreamedLogAsync
16
17
  from apify_client._utils.encoding import encode_key_value_store_record_value
@@ -19,6 +20,7 @@ from apify_client._utils.time import to_seconds
19
20
 
20
21
  if TYPE_CHECKING:
21
22
  import logging
23
+ from collections.abc import AsyncIterator, Iterator
22
24
  from decimal import Decimal
23
25
 
24
26
  from apify_client._literals import GeneralAccess
@@ -32,6 +34,7 @@ if TYPE_CHECKING:
32
34
  RequestQueueClient,
33
35
  RequestQueueClientAsync,
34
36
  )
37
+ from apify_client._resource_clients.dataset import DatasetItemsPage
35
38
  from apify_client.types import Timeout
36
39
 
37
40
 
@@ -469,6 +472,122 @@ class RunClient(ResourceClient):
469
472
 
470
473
  return StatusMessageWatcher(run_client=self, to_logger=to_logger, check_period=check_period)
471
474
 
475
+ def iterate_dataset_items(
476
+ self,
477
+ *,
478
+ offset: int | None = None,
479
+ limit: int | None = None,
480
+ clean: bool | None = None,
481
+ fields: list[str] | None = None,
482
+ omit: list[str] | None = None,
483
+ unwind: list[str] | None = None,
484
+ skip_empty: bool | None = None,
485
+ skip_hidden: bool | None = None,
486
+ chunk_size: int | None = None,
487
+ poll_interval: timedelta = timedelta(seconds=5),
488
+ timeout: Timeout = 'long',
489
+ ) -> Iterator[dict]:
490
+ """Iterate over the items of the run's default dataset while the run is still producing them.
491
+
492
+ While the run has not finished, each poll yields the rows below the dataset's `item_count` and then waits up to
493
+ `poll_interval` for the run to finish, so the last rows are read as soon as it does. Each page is requested with
494
+ a `limit` that ends at `item_count`, so it covers exactly the rows it asks for, whatever the filters or `unwind`
495
+ do to the items. `item_count` lags a few seconds behind the pushed items, so once the run reaches a terminal
496
+ status, the rows past it are read a page at a time until none are left, and the iterator returns. On a
497
+ `last_run()` client, the iterator sticks to the run that its first request resolves to.
498
+
499
+ https://docs.apify.com/api/v2#/reference/datasets/item-collection/get-items
500
+
501
+ Args:
502
+ offset: Number of items that should be skipped at the start. The default value is 0.
503
+ limit: Maximum number of dataset rows to scan. Fewer items are yielded when filters drop some, more
504
+ when `unwind` splits a row into several. By default there is no limit.
505
+ clean: If True, returns only non-empty items and skips hidden fields (i.e. fields starting with
506
+ the # character). The clean parameter is just a shortcut for skip_hidden=True and skip_empty=True
507
+ parameters.
508
+ fields: A list of fields which should be picked from the items, only these fields will remain in
509
+ the resulting record objects.
510
+ omit: A list of fields which should be omitted from the items.
511
+ unwind: A list of fields which should be unwound, in order which they should be processed. Each field
512
+ should be either an array or an object. If the field is an array then every element of the array
513
+ will become a separate record and merged with parent object. If the unwound field is an object then
514
+ it is merged with the parent object.
515
+ skip_empty: If True, then empty items are skipped from the output.
516
+ skip_hidden: If True, then hidden fields are skipped from the output, i.e. fields starting with
517
+ the # character.
518
+ chunk_size: Maximum number of dataset rows requested per API call.
519
+ poll_interval: How long to wait for the run to finish between polls.
520
+ timeout: Timeout for each API HTTP request.
521
+
522
+ Yields:
523
+ An item from the dataset.
524
+ """
525
+ page_size = chunk_size or DEFAULT_CHUNK_SIZE
526
+ position = offset or 0
527
+ end = position + limit if limit else None
528
+
529
+ run = self.get(timeout=timeout)
530
+ # A `last_run()` client resolves `runs/last` per request, so a newer run would swap the dataset mid-iteration.
531
+ run_client = (
532
+ self._client_registry.run_client(
533
+ resource_id=run.id,
534
+ base_url=self._api_base_url,
535
+ public_base_url=self._public_base_url,
536
+ http_client=self._http_client,
537
+ client_registry=self._client_registry,
538
+ )
539
+ if run is not None and run.id != self._resource_id
540
+ else self
541
+ )
542
+ dataset_client = run_client.dataset()
543
+
544
+ def list_page(page_offset: int, page_limit: int) -> DatasetItemsPage:
545
+ return dataset_client.list_items(
546
+ offset=page_offset,
547
+ limit=page_limit,
548
+ clean=clean,
549
+ fields=fields,
550
+ omit=omit,
551
+ unwind=unwind,
552
+ skip_empty=skip_empty,
553
+ skip_hidden=skip_hidden,
554
+ timeout=timeout,
555
+ )
556
+
557
+ while True:
558
+ is_finished = run is None or run.status in _TERMINAL_STATUSES
559
+ dataset = dataset_client.get(timeout=timeout)
560
+ item_count = dataset.item_count if dataset else 0
561
+ if end is not None:
562
+ item_count = min(item_count, end)
563
+
564
+ while position < item_count:
565
+ page_limit = min(page_size, item_count - position)
566
+ page = list_page(position, page_limit)
567
+ yield from page.items
568
+ position += page_limit
569
+
570
+ if end is not None and position >= end:
571
+ return
572
+ if is_finished:
573
+ break
574
+ run = run_client.wait_for_finish(wait_duration=poll_interval, timeout=timeout)
575
+
576
+ while True:
577
+ page_limit = min(page_size, end - position) if end is not None else page_size
578
+ page = list_page(position, page_limit)
579
+ yield from page.items
580
+ # Only an empty page marks the end, as filters can shorten a full one. A page that `clean`, `skip_empty` or
581
+ # `unwind` emptied past a lagging `item_count` reports no scanned rows either, so a plain read checks.
582
+ if not page.count and (
583
+ not (clean or skip_empty or unwind)
584
+ or not dataset_client.list_items(offset=position, limit=1, timeout=timeout).items
585
+ ):
586
+ return
587
+ position += page_limit
588
+ if end is not None and position >= end:
589
+ return
590
+
472
591
 
473
592
  @docs_group('Resource clients')
474
593
  class RunClientAsync(ResourceClientAsync):
@@ -903,3 +1022,121 @@ class RunClientAsync(ResourceClientAsync):
903
1022
  to_logger = create_redirect_logger(f'apify.{name}')
904
1023
 
905
1024
  return StatusMessageWatcherAsync(run_client=self, to_logger=to_logger, check_period=check_period)
1025
+
1026
+ async def iterate_dataset_items(
1027
+ self,
1028
+ *,
1029
+ offset: int | None = None,
1030
+ limit: int | None = None,
1031
+ clean: bool | None = None,
1032
+ fields: list[str] | None = None,
1033
+ omit: list[str] | None = None,
1034
+ unwind: list[str] | None = None,
1035
+ skip_empty: bool | None = None,
1036
+ skip_hidden: bool | None = None,
1037
+ chunk_size: int | None = None,
1038
+ poll_interval: timedelta = timedelta(seconds=5),
1039
+ timeout: Timeout = 'long',
1040
+ ) -> AsyncIterator[dict]:
1041
+ """Iterate over the items of the run's default dataset while the run is still producing them.
1042
+
1043
+ While the run has not finished, each poll yields the rows below the dataset's `item_count` and then waits up to
1044
+ `poll_interval` for the run to finish, so the last rows are read as soon as it does. Each page is requested with
1045
+ a `limit` that ends at `item_count`, so it covers exactly the rows it asks for, whatever the filters or `unwind`
1046
+ do to the items. `item_count` lags a few seconds behind the pushed items, so once the run reaches a terminal
1047
+ status, the rows past it are read a page at a time until none are left, and the iterator returns. On a
1048
+ `last_run()` client, the iterator sticks to the run that its first request resolves to.
1049
+
1050
+ https://docs.apify.com/api/v2#/reference/datasets/item-collection/get-items
1051
+
1052
+ Args:
1053
+ offset: Number of items that should be skipped at the start. The default value is 0.
1054
+ limit: Maximum number of dataset rows to scan. Fewer items are yielded when filters drop some, more
1055
+ when `unwind` splits a row into several. By default there is no limit.
1056
+ clean: If True, returns only non-empty items and skips hidden fields (i.e. fields starting with
1057
+ the # character). The clean parameter is just a shortcut for skip_hidden=True and skip_empty=True
1058
+ parameters.
1059
+ fields: A list of fields which should be picked from the items, only these fields will remain in
1060
+ the resulting record objects.
1061
+ omit: A list of fields which should be omitted from the items.
1062
+ unwind: A list of fields which should be unwound, in order which they should be processed. Each field
1063
+ should be either an array or an object. If the field is an array then every element of the array
1064
+ will become a separate record and merged with parent object. If the unwound field is an object then
1065
+ it is merged with the parent object.
1066
+ skip_empty: If True, then empty items are skipped from the output.
1067
+ skip_hidden: If True, then hidden fields are skipped from the output, i.e. fields starting with
1068
+ the # character.
1069
+ chunk_size: Maximum number of dataset rows requested per API call.
1070
+ poll_interval: How long to wait for the run to finish between polls.
1071
+ timeout: Timeout for each API HTTP request.
1072
+
1073
+ Yields:
1074
+ An item from the dataset.
1075
+ """
1076
+ page_size = chunk_size or DEFAULT_CHUNK_SIZE
1077
+ position = offset or 0
1078
+ end = position + limit if limit else None
1079
+
1080
+ run = await self.get(timeout=timeout)
1081
+ # A `last_run()` client resolves `runs/last` per request, so a newer run would swap the dataset mid-iteration.
1082
+ run_client = (
1083
+ self._client_registry.run_client(
1084
+ resource_id=run.id,
1085
+ base_url=self._api_base_url,
1086
+ public_base_url=self._public_base_url,
1087
+ http_client=self._http_client,
1088
+ client_registry=self._client_registry,
1089
+ )
1090
+ if run is not None and run.id != self._resource_id
1091
+ else self
1092
+ )
1093
+ dataset_client = run_client.dataset()
1094
+
1095
+ async def list_page(page_offset: int, page_limit: int) -> DatasetItemsPage:
1096
+ return await dataset_client.list_items(
1097
+ offset=page_offset,
1098
+ limit=page_limit,
1099
+ clean=clean,
1100
+ fields=fields,
1101
+ omit=omit,
1102
+ unwind=unwind,
1103
+ skip_empty=skip_empty,
1104
+ skip_hidden=skip_hidden,
1105
+ timeout=timeout,
1106
+ )
1107
+
1108
+ while True:
1109
+ is_finished = run is None or run.status in _TERMINAL_STATUSES
1110
+ dataset = await dataset_client.get(timeout=timeout)
1111
+ item_count = dataset.item_count if dataset else 0
1112
+ if end is not None:
1113
+ item_count = min(item_count, end)
1114
+
1115
+ while position < item_count:
1116
+ page_limit = min(page_size, item_count - position)
1117
+ page = await list_page(position, page_limit)
1118
+ for item in page.items:
1119
+ yield item
1120
+ position += page_limit
1121
+
1122
+ if end is not None and position >= end:
1123
+ return
1124
+ if is_finished:
1125
+ break
1126
+ run = await run_client.wait_for_finish(wait_duration=poll_interval, timeout=timeout)
1127
+
1128
+ while True:
1129
+ page_limit = min(page_size, end - position) if end is not None else page_size
1130
+ page = await list_page(position, page_limit)
1131
+ for item in page.items:
1132
+ yield item
1133
+ # Only an empty page marks the end, as filters can shorten a full one. A page that `clean`, `skip_empty` or
1134
+ # `unwind` emptied past a lagging `item_count` reports no scanned rows either, so a plain read checks.
1135
+ if not page.count and (
1136
+ not (clean or skip_empty or unwind)
1137
+ or not (await dataset_client.list_items(offset=position, limit=1, timeout=timeout)).items
1138
+ ):
1139
+ return
1140
+ position += page_limit
1141
+ if end is not None and position >= end:
1142
+ return
File without changes
File without changes