Skip to content
1 change: 1 addition & 0 deletions docs/02_concepts/07_convenience_methods.mdx
Original file line number Diff line number Diff line change
Expand Up @@ -18,6 +18,7 @@ The Apify client provides several convenience methods to handle actions that the
- <ApiLink to="class/ActorClient#call">`ActorClient.call`</ApiLink> - Starts an Actor and waits for it to finish, handling network timeouts internally. Waits indefinitely by default, or up to the specified `wait_duration`.
- <ApiLink to="class/ActorClient#start">`ActorClient.start`</ApiLink> - Starts an Actor and immediately returns the Run object without waiting for it to finish.
- <ApiLink to="class/RunClient#wait_for_finish">`RunClient.wait_for_finish`</ApiLink> - Waits for an already-started run to reach a terminal status.
- <ApiLink to="class/RunClient#iterate_dataset_items">`RunClient.iterate_dataset_items`</ApiLink> - Yields the items of the run's default dataset as the run pushes them, and returns once the run has finished and every item is read.

Additionally, storage-related resources offer flexible options for data retrieval:

Expand Down
2 changes: 2 additions & 0 deletions docs/02_concepts/08_pagination.mdx
Original file line number Diff line number Diff line change
Expand Up @@ -83,3 +83,5 @@ The next example uses `iterate_items` on a dataset client to stream items past a
</CodeBlock>
</TabItem>
</Tabs>

To read the items of a run that's still going, use <ApiLink to="class/RunClient#iterate_dataset_items">`RunClient.iterate_dataset_items`</ApiLink>. Its `offset` and `limit` also count dataset rows. For details, see [Retrieve Actor data](/api/client/python/docs/guides/retrieve-actor-data#read-items-while-the-run-is-going).
30 changes: 30 additions & 0 deletions docs/03_guides/03_retrieve_actor_data.mdx
Original file line number Diff line number Diff line change
Expand Up @@ -8,6 +8,10 @@ import Tabs from '@theme/Tabs';
import TabItem from '@theme/TabItem';
import CodeBlock from '@theme/CodeBlock';

import ApiLink from '@theme/ApiLink';

import LiveItemsAsyncExample from '!!raw-loader!./code/03_live_items_async.py';
import LiveItemsSyncExample from '!!raw-loader!./code/03_live_items_sync.py';
import RetrieveAsyncExample from '!!raw-loader!./code/03_retrieve_async.py';
import RetrieveSyncExample from '!!raw-loader!./code/03_retrieve_sync.py';

Expand All @@ -27,3 +31,29 @@ The following example shows how to fetch datasets from an Actor's runs, paginate
</CodeBlock>
</TabItem>
</Tabs>

## Read items while the run is going

To process a run's output before the run finishes, use <ApiLink to="class/RunClient#iterate_dataset_items">`RunClient.iterate_dataset_items`</ApiLink>. It yields the items of the run's default dataset shortly after the run pushes them, and it returns once the run has finished and every item is read.

Note that:

- Between polls, the iterator waits up to `poll_interval` (5 seconds by default) for the run to finish. Once it finishes, the iterator reads the remaining items right away. An explicit `timeout` has to leave room for that wait.
- The item options are the same as in <ApiLink to="class/DatasetClient#iterate_items">`DatasetClient.iterate_items`</ApiLink>, except for `desc` and `signature`.
- `offset` and `limit` count dataset rows, not the items you get back. With `clean`, `skip_empty` or `unwind`, the iterator can yield fewer or more items than `limit`.
- A run that's `ABORTING` or `TIMING-OUT` can still push items, so the iterator keeps polling until the run reaches a terminal status.

The following example starts an Actor and prints its items as the run produces them:

<Tabs>
<TabItem value="AsyncExample" label="Async client" default>
<CodeBlock className="language-python">
{LiveItemsAsyncExample}
</CodeBlock>
</TabItem>
<TabItem value="SyncExample" label="Sync client">
<CodeBlock className="language-python">
{LiveItemsSyncExample}
</CodeBlock>
</TabItem>
</Tabs>
22 changes: 22 additions & 0 deletions docs/03_guides/code/03_live_items_async.py
Original file line number Diff line number Diff line change
@@ -0,0 +1,22 @@
import asyncio

from apify_client import ApifyClientAsync

TOKEN = 'MY-APIFY-TOKEN'


async def main() -> None:
apify_client = ApifyClientAsync(TOKEN)

# Start the Actor without waiting for it to finish
actor_client = apify_client.actor('username/actor-name')
run = await actor_client.start(run_input={'query': 'web scraping'})

# Each item arrives shortly after the run pushes it. The loop ends once the run
# has finished and every item is read.
async for item in apify_client.run(run.id).iterate_dataset_items(skip_empty=True):
print(item)


if __name__ == '__main__':
asyncio.run(main())
20 changes: 20 additions & 0 deletions docs/03_guides/code/03_live_items_sync.py
Original file line number Diff line number Diff line change
@@ -0,0 +1,20 @@
from apify_client import ApifyClient

TOKEN = 'MY-APIFY-TOKEN'


def main() -> None:
apify_client = ApifyClient(TOKEN)

# Start the Actor without waiting for it to finish
actor_client = apify_client.actor('username/actor-name')
run = actor_client.start(run_input={'query': 'web scraping'})

# Each item arrives shortly after the run pushes it. The loop ends once the run
# has finished and every item is read.
for item in apify_client.run(run.id).iterate_dataset_items(skip_empty=True):
print(item)


if __name__ == '__main__':
main()
12 changes: 11 additions & 1 deletion src/apify_client/_resource_clients/_resource_client.py
Original file line number Diff line number Diff line change
Expand Up @@ -42,6 +42,7 @@ def __init__(
client_registry: Any,
resource_id: str | None = None,
params: dict | None = None,
api_base_url: str | None = None,
) -> None:
"""Initialize the resource client.

Expand All @@ -53,11 +54,13 @@ def __init__(
client_registry: Bundle of client classes for dependency injection.
resource_id: Optional resource ID for single-resource clients.
params: Optional default parameters for all requests.
api_base_url: Base URL of the API itself, for clients of top-level resources. Defaults to `base_url`.
"""
if resource_path.endswith('/'):
raise ValueError('resource_path must not end with "/"')

self._base_url = base_url
self._api_base_url = api_base_url or base_url
self._public_base_url = public_base_url
self._http_client = http_client
self._default_params = params or {}
Expand All @@ -82,11 +85,12 @@ def _resource_url(self) -> str:
def _base_client_kwargs(self) -> dict[str, Any]:
"""Base kwargs for creating nested/child clients.

Returns dict with base_url, public_base_url, http_client, and client_registry. Caller adds
Returns dict with base_url, api_base_url, public_base_url, http_client, and client_registry. Caller adds
resource_path, resource_id, and params as needed.
"""
return {
'base_url': self._resource_url,
'api_base_url': self._api_base_url,
'public_base_url': self._public_base_url,
'http_client': self._http_client,
'client_registry': self._client_registry,
Expand Down Expand Up @@ -197,6 +201,7 @@ def __init__(
client_registry: ClientRegistry,
resource_id: str | None = None,
params: dict | None = None,
api_base_url: str | None = None,
) -> None:
"""Initialize the resource client.

Expand All @@ -208,6 +213,7 @@ def __init__(
client_registry: Bundle of client classes for dependency injection.
resource_id: Optional resource ID for single-resource clients.
params: Optional default parameters for all requests.
api_base_url: Base URL of the API itself, for clients of top-level resources. Defaults to `base_url`.
"""
super().__init__(
base_url=base_url,
Expand All @@ -217,6 +223,7 @@ def __init__(
client_registry=client_registry,
resource_id=resource_id,
params=params,
api_base_url=api_base_url,
)

def _get(self, *, timeout: Timeout) -> dict | None:
Expand Down Expand Up @@ -389,6 +396,7 @@ def __init__(
client_registry: ClientRegistryAsync,
resource_id: str | None = None,
params: dict | None = None,
api_base_url: str | None = None,
) -> None:
"""Initialize the resource client.

Expand All @@ -400,6 +408,7 @@ def __init__(
client_registry: Bundle of client classes for dependency injection.
resource_id: Optional resource ID for single-resource clients.
params: Optional default parameters for all requests.
api_base_url: Base URL of the API itself, for clients of top-level resources. Defaults to `base_url`.
"""
super().__init__(
base_url=base_url,
Expand All @@ -409,6 +418,7 @@ def __init__(
client_registry=client_registry,
resource_id=resource_id,
params=params,
api_base_url=api_base_url,
)

async def _get(self, *, timeout: Timeout) -> dict | None:
Expand Down
Loading
Loading