1from __future__ import annotations
2
3import requests
4from bs4 import BeautifulSoup
5
6from apify import Actor
7
8
9def extract_title_and_meta(url: str) -> dict:
10 try:
11 headers = {
12 'User-Agent': 'Mozilla/5.0 (Windows NT 10.0; Win64; x64) '
13 'AppleWebKit/537.36 (KHTML, like Gecko) '
14 'Chrome/91.0.4472.124 Safari/537.36'
15 }
16 r = requests.get(url, headers=headers, timeout=10)
17 r.raise_for_status()
18 soup = BeautifulSoup(r.text, 'html.parser')
19
20 title = soup.title.string if soup.title else ''
21
22 meta_tag = soup.find('meta', attrs={'name': 'description'})
23 if meta_tag and meta_tag.get('content'):
24 meta_desc = meta_tag.get('content', '')
25 else:
26 meta_tag = soup.find('meta', attrs={'property': 'og:description'})
27 meta_desc = meta_tag.get('content', '') if meta_tag else ''
28
29 return {
30 'url': url,
31 'title': (title or '').strip(),
32 'meta_description': (meta_desc or '').strip(),
33 'error': None,
34 }
35 except Exception as e:
36 return {
37 'url': url,
38 'title': None,
39 'meta_description': None,
40 'error': str(e),
41 }
42
43
44async def main() -> None:
45 async with Actor:
46 actor_input = await Actor.get_input() or {}
47
48 urls = actor_input.get('urls')
49 if not urls:
50 single_url = actor_input.get('url')
51 urls = [single_url] if single_url else []
52
53
54
55 normalized_urls = []
56 for item in urls:
57 if isinstance(item, str):
58 normalized_urls.append(item)
59 elif isinstance(item, dict) and item.get('url'):
60 normalized_urls.append(item['url'])
61
62 if not normalized_urls:
63 Actor.log.warning('No URLs provided in input. Nothing to do.')
64 return
65
66 for url in normalized_urls:
67 Actor.log.info(f'Extracting title/meta from {url}')
68 result = extract_title_and_meta(url)
69 await Actor.push_data(result)
70
71
72if __name__ == '__main__':
73 import asyncio
74 asyncio.run(main())