1from apify_client import ApifyClient
2
3
4
5client = ApifyClient("<YOUR_API_TOKEN>")
6
7
8run_input = {
9 "fileUrls": [
10 "https://arxiv.org/pdf/2305.10601",
11 "https://arxiv.org/pdf/1706.03762",
12 "https://raw.githubusercontent.com/microsoft/markitdown/main/packages/markitdown/tests/test_files/test.docx",
13 "https://raw.githubusercontent.com/microsoft/markitdown/main/packages/markitdown/tests/test_files/test.xlsx",
14 "https://raw.githubusercontent.com/microsoft/markitdown/main/packages/markitdown/tests/test_files/test.pptx",
15 "https://raw.githubusercontent.com/microsoft/markitdown/main/packages/markitdown/tests/test_files/test.json",
16 "https://raw.githubusercontent.com/microsoft/markitdown/main/README.md",
17 "https://raw.githubusercontent.com/apify/apify-sdk-python/master/README.md",
18 "https://people.sc.fsu.edu/~jburkardt/data/csv/airtravel.csv",
19 "https://www.ietf.org/rfc/rfc2119.txt",
20 "https://www.gutenberg.org/cache/epub/11/pg11.txt",
21 "https://example.com",
22 ],
23 "includePageBreaks": False,
24 "truncateChars": 0,
25}
26
27
28run = client.actor("ntriqpro/pdf-to-markdown").call(run_input=run_input)
29
30
31print(f"💾 Check your data here: https://console.apify.com/storage/datasets/{run.default_dataset_id}")
32for item in client.dataset(run.default_dataset_id).iterate_items():
33 print(item)
34
35