Retrieve Actor data
Actor output data is stored in datasets, which can be retrieved from individual Actor runs. Dataset items support pagination for efficient retrieval, and multiple datasets can be merged into a single dataset for further analysis. This merged dataset can then be exported into various formats such as CSV, JSON, XLSX, or XML. Additionally, integrations provide powerful tools to automate data workflows.
The following example shows how to fetch datasets from an Actor's runs, paginate through their items, and merge them into a single dataset for unified analysis:
- Async client
- Sync client
import asyncio
from apify_client import ApifyClientAsync
TOKEN = 'MY-APIFY-TOKEN'
async def main() -> None:
# Client initialization with the API token
apify_client = ApifyClientAsync(token=TOKEN)
actor_client = apify_client.actor('apify/instagram-hashtag-scraper')
runs_client = actor_client.runs()
# See pagination to understand how to get more datasets
actor_datasets = await runs_client.list(limit=20)
datasets_client = apify_client.datasets()
merging_dataset = await datasets_client.get_or_create(name='merge-dataset')
for dataset_item in actor_datasets.items:
# Dataset items can be handled here. Dataset items can be paginated
dataset_client = apify_client.dataset(dataset_item.id)
dataset_items = await dataset_client.list_items(limit=1000)
# Items can be pushed to single dataset
merging_dataset_client = apify_client.dataset(merging_dataset.id)
await merging_dataset_client.push_items(dataset_items.items)
# ...
if __name__ == '__main__':
asyncio.run(main())
from apify_client import ApifyClient
TOKEN = 'MY-APIFY-TOKEN'
def main() -> None:
# Client initialization with the API token
apify_client = ApifyClient(token=TOKEN)
actor_client = apify_client.actor('apify/instagram-hashtag-scraper')
runs_client = actor_client.runs()
# See pagination to understand how to get more datasets
actor_datasets = runs_client.list(limit=20)
datasets_client = apify_client.datasets()
merging_dataset = datasets_client.get_or_create(name='merge-dataset')
for dataset_item in actor_datasets.items:
# Dataset items can be handled here. Dataset items can be paginated
dataset_client = apify_client.dataset(dataset_item.id)
dataset_items = dataset_client.list_items(limit=1000)
# Items can be pushed to single dataset
merging_dataset_client = apify_client.dataset(merging_dataset.id)
merging_dataset_client.push_items(dataset_items.items)
# ...
if __name__ == '__main__':
main()
Read items while the run is going
To process a run's output before the run finishes, use RunClient.iterate_dataset_items. It yields the items of the run's default dataset shortly after the run pushes them, and it returns once the run has finished and every item is read.
Note that:
- Between polls, the iterator waits up to
poll_interval(5 seconds by default) for the run to finish. Once it finishes, the iterator reads the remaining items right away. An explicittimeouthas to leave room for that wait. - The item options are the same as in
DatasetClient.iterate_items, except fordescandsignature. offsetandlimitcount dataset rows, not the items you get back. Withclean,skip_emptyorunwind, the iterator can yield fewer or more items thanlimit.- A run that's
ABORTINGorTIMING-OUTcan still push items, so the iterator keeps polling until the run reaches a terminal status.
The following example starts an Actor and prints its items as the run produces them:
- Async client
- Sync client
import asyncio
from apify_client import ApifyClientAsync
TOKEN = 'MY-APIFY-TOKEN'
async def main() -> None:
apify_client = ApifyClientAsync(TOKEN)
# Start the Actor without waiting for it to finish
actor_client = apify_client.actor('username/actor-name')
run = await actor_client.start(run_input={'query': 'web scraping'})
# Each item arrives shortly after the run pushes it. The loop ends once the run
# has finished and every item is read.
async for item in apify_client.run(run.id).iterate_dataset_items(skip_empty=True):
print(item)
if __name__ == '__main__':
asyncio.run(main())
from apify_client import ApifyClient
TOKEN = 'MY-APIFY-TOKEN'
def main() -> None:
apify_client = ApifyClient(TOKEN)
# Start the Actor without waiting for it to finish
actor_client = apify_client.actor('username/actor-name')
run = actor_client.start(run_input={'query': 'web scraping'})
# Each item arrives shortly after the run pushes it. The loop ends once the run
# has finished and every item is read.
for item in apify_client.run(run.id).iterate_dataset_items(skip_empty=True):
print(item)
if __name__ == '__main__':
main()