1"""
2This module defines the `main()` coroutine for the Apify Actor, executed from the `__main__.py` file.
3
4Feel free to modify this file to suit your specific needs.
5
6To build Apify Actors, utilize the Apify SDK toolkit, read more at the official documentation:
7https://docs.apify.com/sdk/python
8"""
9
10from urllib.parse import urljoin
11from selenium.webdriver.common.action_chains import ActionChains
12from apify import Actor
13import re
14from urllib.parse import urljoin
15from bs4 import BeautifulSoup
16import requests
17import requests.exceptions
18import time
19import random
20from .datascrapify import StartProcess, parse_proxy_url,on_aborting_event
21import asyncio
22from urllib.parse import urlparse, urlsplit, urlunsplit, parse_qs, urlencode
23import os
24
25
26
27
28
29import datetime
30
31async def check_free_trial(user_id: str, is_paid_user: bool = False):
32 """
33 Returns True if the user is allowed to use the Actor.
34
35 Paid users: always allowed.
36 Free users: allowed for 24 hours from first use.
37 """
38
39
40 if is_paid_user:
41 return True
42
43 store = await Actor.open_key_value_store(name="user-trials-linkedinemailscrapers")
44
45 trial_key = f"trial_start_{user_id}"
46
47 now = datetime.datetime.now(datetime.timezone.utc)
48
49 first_use = await store.get_value(trial_key)
50
51
52 if not first_use:
53 await store.set_value(
54 trial_key,
55 now.isoformat()
56 )
57
58 Actor.log.info(
59 f"Free trial started for user {user_id}. "
60 f"Trial valid for 24 hours."
61 )
62
63 return True
64
65
66 first_use_date = datetime.datetime.fromisoformat(first_use)
67
68
69 if first_use_date.tzinfo is None:
70 first_use_date = first_use_date.replace(
71 tzinfo=datetime.timezone.utc
72 )
73
74 elapsed = now - first_use_date
75
76 if elapsed < datetime.timedelta(hours=24):
77 remaining = datetime.timedelta(hours=24) - elapsed
78
79 hours = int(remaining.total_seconds() // 3600)
80 minutes = int((remaining.total_seconds() % 3600) // 60)
81
82 Actor.log.info(
83 f"Free trial active. "
84 f"Remaining: {hours}h {minutes}m"
85 )
86
87 return True
88
89 Actor.log.warning(
90 f"Free trial expired for user {user_id}"
91 )
92
93 return False
94
95def scrape_contact_emails(link):
96 res = requests.get(link)
97 domain = link.split(".")
98 mailaddr = link
99 soup = BeautifulSoup(res.text,"lxml")
100 links = soup.find_all("a")
101 contact_link = ''
102 final_result = ""
103 try:
104
105 emails = soup.find_all(text=re.compile('.*@'+domain[1]+'.'+domain[2].replace("/","")))
106 emails.sort(key=len)
107 print(emails[0].replace("\n",""))
108 final_result = emails[0]
109 except:
110
111 try:
112 flag = 0
113 for link in links:
114 if "contact" in link.get("href") or "Contact" in link.get("href") or "CONTACT" in link.get("href") or 'contact' in link.text or 'Contact' in link.text or 'CONTACT' in link.text:
115 if len(link.get("href"))>2 and flag<2:
116 flag = flag + 1
117 contact_link = link.get("href")
118
119 except:
120 pass
121
122 domain = domain[0]+"."+domain[1]+"."+domain[2]
123 if(len(contact_link)<len(domain)):
124 domain = domain+contact_link.replace("/","")
125 else:
126 domain = contact_link
127
128 try:
129
130 res = requests.get(domain)
131 soup = BeautifulSoup(res.text,"lxml")
132 emails = soup.find_all(text=re.compile('.*@'+mailaddr[7:].replace("/","")))
133 emails.sort(key=len)
134 try:
135 print(emails[0].replace("\n",""))
136 final_result = emails[0]
137 return final_result
138 except:
139 pass
140 except Exception as e:
141 pass
142
143 return ""
144
145async def main() -> None:
146 """
147 The main coroutine is being executed using `asyncio.run()`, so do not attempt to make a normal function
148 out of it, it will not work. Asynchronous execution is required for communication with Apify platform,
149 and it also enhances performance in the field of web scraping significantly.
150 """
151 async with Actor:
152
153 actor_input = await Actor.get_input() or {}
154 Keyword_val = actor_input.get('Keyword')
155 location_val = actor_input.get('location')
156 social_network_val = actor_input.get('social_network')
157 Country_val = actor_input.get('Country')
158 Email_Type_val = actor_input.get('Email_Type')
159 Other_Email_Type_val = actor_input.get('Other_Email_Type')
160 proxy_settings = actor_input.get('proxySettings')
161 Limit_val =actor_input.get('Limit')
162
163
164 proxy_configuration = await Actor.create_proxy_configuration(groups=['GOOGLE_SERP'])
165
166 proxyurl =await proxy_configuration.new_url()
167 if proxy_configuration and proxy_settings:
168 proxyurl =await proxy_configuration.new_url()
169
170
171
172 if not Keyword_val:
173 Actor.log.info('Please insert keyword')
174 await Actor.push_data({'Email': 'Please insert keyword'})
175 await Actor.exit()
176 return
177
178 if Keyword_val=='TestKeyword':
179 Actor.log.info('Please insert keyword')
180 await Actor.push_data({'Email': 'Please insert Your Keyword'})
181 await Actor.exit()
182 return
183
184 is_paying = os.getenv("APIFY_USER_IS_PAYING")
185 print(is_paying)
186 if is_paying=="1":
187 print('Paid User')
188 else:
189 Limit_val=10
190 is_paid_user=False
191 APIFY_USER_ID = os.getenv("APIFY_USER_ID")
192 allowed = await check_free_trial(user_id=APIFY_USER_ID, is_paid_user=is_paid_user)
193 if not allowed:
194 await Actor.push_data({'Email': 'Free User can run for One Day Only. Please Subcribe'})
195 return
196 await Actor.push_data({'Email': 'Free User can run tool with 1 day with 10 record'})
197
198 '''
199 try:
200 proxyurl=''
201 USE_Proxy=False
202 if proxyurl:
203 USE_Proxy=True
204 my_proxy_settings = parse_proxy_url(proxyurl)
205 # Call the function from the imported module
206 await StartProcess(
207 "Apify_camp_LinkedinEmail_"+username+"_"+isPaying,
208 "ALLINONE",
209 Keyword_val,
210 location_val,
211 social_network_val,
212 Country_val,
213 "Google",
214 USE_Proxy,
215 my_proxy_settings
216 )
217
218 finally:
219 # Code that always executes (e.g., cleanup)
220 print("This block always runs, regardless of exceptions.")
221
222 await Actor.exit();
223 '''
224
225 l1 = ["it","www","gm","fr","sp","uk","al","ag","ar","am","as","au","aj","bg","bo","be","bh","bk","br","bu","ca","ci","ch","co","cs","hr","cu","ez","dk","ec","eg","en","fi","gg","gr","hk","hu","ic","in","id","ir","iz","ei","is","jm","ja","ke","kn","ks","ku","lg","ly","ls","lh","lu","mc","mk","my","mt","mx","md","mn","mj","mo","np","nl","nz","ni","no","pk","we","pm","pa","pe","rp","pl","po","rq","qa","ro","rs","sm","sa","sg","ri","sn","lo","si","sf","sw","sz","sy","tw","th","ts","tu","ua","ae","uy","uz","ve"]
226 l2= ["Italy","United States","Germany","France","Spain","United Kingdom","Albania","Algeria","Argentina","Armenia","Australia","Austria","Azerbaijan","Bangladesh","Belarus","Belgium","Belize","Bosnia and Herzegovina","Brazil","Bulgaria","Canada","Chile","China","Colombia","Costa Rica","Croatia","Cuba","Czechia","Denmark","Ecuador","Egypt","Estonia","Finland","Georgia","Greece","Hong Kong","Hungary","Iceland","India","Indonesia","Iran","Iraq","Ireland","Israel","Jamaica","Japan","Kenya","Korea","Korea, Republic of","Kuwait","Latvia","Libya","Liechtenstein","Lithuania","Luxembourg","Macao","Macedonia","Malaysia","Malta","Mexico","Moldova, Republic of","Monaco","Montenegro","Morocco","Nepal","Netherlands","New Zealand","Nigeria","Norway","Pakistan","Palestine, State of","Panama","Paraguay","Peru","Philippines","Poland","Portugal","Puerto Rico","Qatar","Romania","Russia","San Marino","Saudi Arabia","Senegal","Serbia","Singapore","Slovakia","Slovenia","South Africa","Sweden","Switzerland","Syrian Arab Republic","Taiwan","Thailand","Tunisia","Turkey","Ukraine","United Arab Emirates","Uruguay","Uzbekistan","Venezuela"]
227 select_index=1
228 select_country='United States'
229 for count, ele in enumerate(l1):
230 if(ele==Country_val):
231 select_index=count
232 break
233
234
235
236 for count, ele in enumerate(l2):
237 if(count==select_index):
238 select_country=ele
239 break
240
241 print(select_country)
242
243 concatstring = ""
244 concatstring = concatstring + Keyword_val
245 option = "( @gmail.com OR @hotmail.com OR @yahoo.com)";
246 if Email_Type_val=="1":
247 if not Other_Email_Type_val:
248 Actor.log.info('Please insert Email Type Domain')
249 await Actor.push_data({'Email': 'Please insert Email Type Domain'})
250 await Actor.exit()
251 return
252 if Other_Email_Type_val.find("@") > -1:
253 option = " ( " + Other_Email_Type_val + " )"
254 else:
255 option = " ( @" + Other_Email_Type_val + " )"
256 concatstring = concatstring + option
257 if location_val:
258 concatstring = concatstring+ " in "+ location_val
259
260
261
262 if social_network_val:
263 concatstring = concatstring + " site:"
264
265 if social_network_val == "linkedin.com/" or social_network_val == "pinterest.com/" :
266 concatstring = concatstring + Country_val + ".";
267
268
269
270
271 if social_network_val == "amazon.com/" :
272 if Country_val=='gm':
273 Country_val='de'
274 elif Country_val=='sp':
275 Country_val='es'
276 elif Country_val=='fr':
277 Country_val='fr'
278 elif Country_val=='uk':
279 Country_val='co.uk'
280 elif Country_val=='as':
281 Country_val='com.au'
282 elif Country_val=='www':
283 Country_val='com'
284 elif Country_val=='in':
285 Country_val='in'
286 elif Country_val=='be':
287 Country_val='com.be'
288 elif Country_val=='br':
289 Country_val='com.br'
290 elif Country_val=='ca':
291 Country_val='ca'
292 elif Country_val=='ch':
293 Country_val='cn'
294 elif Country_val=='eg':
295 Country_val='eg'
296 elif Country_val=='it':
297 Country_val='it'
298 elif Country_val=='ja':
299 Country_val='co.jp'
300 elif Country_val=='mx':
301 Country_val='com.mx'
302 elif Country_val=='nl':
303 Country_val='nl'
304 elif Country_val=='pl':
305 Country_val='pl'
306 elif Country_val=='sa':
307 Country_val='sa'
308 elif Country_val=='sn':
309 Country_val='sg'
310 elif Country_val=='sw':
311 Country_val='se'
312 elif Country_val=='tu':
313 Country_val='com.tr'
314 elif Country_val=='ae':
315 Country_val='ae'
316 elif Country_val=='ae':
317 Country_val='ae'
318 else :
319 Country_val=='com'
320
321 social_network_val=social_network_val.replace('.com','.'+Country_val)
322
323
324 concatstring = concatstring + "" + social_network_val + "";
325
326 SearchEngine='Google'
327 desktop_user_agents = ["Mozilla/5.0 (Windows NT 10.0; Win64; x64) AppleWebKit/537.36 (KHTML, like Gecko) Chrome/123.0.0.0 Safari/537.36",
328 "Mozilla/5.0 (Windows NT 10.0; Win64; x64) AppleWebKit/537.36 (KHTML, like Gecko) Chrome/122.0.0.0 Safari/537.36",
329 "Mozilla/5.0 (Windows NT 10.0; Win64; x64) AppleWebKit/537.36 (KHTML, like Gecko) Chrome/121.0.6167.139 Safari/537.36",
330 "Mozilla/5.0 (Windows NT 6.1; Win64; x64) AppleWebKit/537.36 (KHTML, like Gecko) Chrome/120.0.0.0 Safari/537.36",
331
332 "Mozilla/5.0 (Windows NT 10.0; Win64; x64; rv:125.0) Gecko/20100101 Firefox/125.0",
333 "Mozilla/5.0 (Windows NT 10.0; rv:124.0) Gecko/20100101 Firefox/124.0",
334 "Mozilla/5.0 (Windows NT 6.1; Win64; x64; rv:122.0) Gecko/20100101 Firefox/122.0",
335
336 "Mozilla/5.0 (Windows NT 10.0; Win64; x64) AppleWebKit/537.36 (KHTML, like Gecko) Chrome/123.0.6312.86 Safari/537.36 Edg/123.0.2420.65",
337 "Mozilla/5.0 (Windows NT 10.0; Win64; x64) AppleWebKit/537.36 (KHTML, like Gecko) Chrome/122.0.6261.112 Safari/537.36 Edg/122.0.2365.66",
338
339 "Mozilla/5.0 (Windows NT 10.0; Win64; x64) AppleWebKit/537.36 (KHTML, like Gecko) Chrome/122.0.0.0 Safari/537.36 OPR/96.0.0.0",
340
341 "Mozilla/5.0 (Macintosh; Intel Mac OS X 13_5_2) AppleWebKit/537.36 (KHTML, like Gecko) Chrome/122.0.6261.129 Safari/537.36",
342 "Mozilla/5.0 (Macintosh; Intel Mac OS X 12_6_3) AppleWebKit/537.36 (KHTML, like Gecko) Chrome/121.0.0.0 Safari/537.36",
343
344 "Mozilla/5.0 (Macintosh; Intel Mac OS X 13.5; rv:124.0) Gecko/20100101 Firefox/124.0",
345 "Mozilla/5.0 (Macintosh; Intel Mac OS X 10.15; rv:123.0) Gecko/20100101 Firefox/123.0",
346
347 "Mozilla/5.0 (Macintosh; Intel Mac OS X 13_4) AppleWebKit/605.1.15 (KHTML, like Gecko) Version/16.4 Safari/605.1.15",
348 "Mozilla/5.0 (Macintosh; Intel Mac OS X 12_3_1) AppleWebKit/605.1.15 (KHTML, like Gecko) Version/15.4 Safari/605.1.15",
349
350 "Mozilla/5.0 (X11; Linux x86_64) AppleWebKit/537.36 (KHTML, like Gecko) Chrome/121.0.6167.184 Safari/537.36",
351 "Mozilla/5.0 (X11; Linux x86_64) AppleWebKit/537.36 (KHTML, like Gecko) Chrome/120.0.0.0 Safari/537.36",
352
353 "Mozilla/5.0 (X11; Linux x86_64; rv:125.0) Gecko/20100101 Firefox/125.0",
354 "Mozilla/5.0 (X11; Linux x86_64; rv:124.0) Gecko/20100101 Firefox/124.0",
355 "Mozilla/5.0 (X11; Ubuntu; Linux x86_64; rv:123.0) Gecko/20100101 Firefox/123.0",
356
357 "Mozilla/5.0 (Macintosh; Intel Mac OS X 13_2_1) AppleWebKit/605.1.15 (KHTML, like Gecko) Version/16.3 Safari/605.1.15",
358
359 "Mozilla/5.0 (Macintosh; Intel Mac OS X 13_4_1) AppleWebKit/537.36 (KHTML, like Gecko) Chrome/123.0.6312.86 Safari/537.36 Edg/123.0.2420.65",
360
361 "Mozilla/5.0 (Macintosh; Intel Mac OS X 13_2_1) AppleWebKit/537.36 (KHTML, like Gecko) Chrome/123.0.0.0 Safari/537.36 OPR/96.0.0.0"
362 ]
363
364 mobile_user_agents = [
365
366 "Mozilla/5.0 (Linux; Android 13; Pixel 7 Pro) AppleWebKit/537.36 (KHTML, like Gecko) Chrome/123.0.6312.105 Mobile Safari/537.36",
367 "Mozilla/5.0 (Linux; Android 12; Pixel 6) AppleWebKit/537.36 (KHTML, like Gecko) Chrome/122.0.0.0 Mobile Safari/537.36",
368 "Mozilla/5.0 (Linux; Android 11; Pixel 4a) AppleWebKit/537.36 (KHTML, like Gecko) Chrome/121.0.6167.140 Mobile Safari/537.36",
369
370
371 "Mozilla/5.0 (Linux; Android 13; SAMSUNG SM-S911B) AppleWebKit/537.36 (KHTML, like Gecko) SamsungBrowser/24.0 Chrome/123.0.0.0 Mobile Safari/537.36",
372 "Mozilla/5.0 (Linux; Android 12; SAMSUNG SM-A525F) AppleWebKit/537.36 (KHTML, like Gecko) SamsungBrowser/23.0 Chrome/121.0.0.0 Mobile Safari/537.36",
373
374
375 "Mozilla/5.0 (Linux; Android 13; M2101K6G) AppleWebKit/537.36 (KHTML, like Gecko) Chrome/123.0.6312.105 Mobile Safari/537.36",
376
377
378 "Mozilla/5.0 (iPhone; CPU iPhone OS 17_4 like Mac OS X) AppleWebKit/605.1.15 (KHTML, like Gecko) Version/17.0 Mobile/15E148 Safari/604.1",
379 "Mozilla/5.0 (iPhone; CPU iPhone OS 16_6 like Mac OS X) AppleWebKit/605.1.15 (KHTML, like Gecko) Version/16.0 Mobile/15E148 Safari/604.1",
380 "Mozilla/5.0 (iPhone; CPU iPhone OS 15_5 like Mac OS X) AppleWebKit/605.1.15 (KHTML, like Gecko) Version/15.5 Mobile/15E148 Safari/604.1",
381
382
383 "Mozilla/5.0 (iPad; CPU OS 17_4 like Mac OS X) AppleWebKit/605.1.15 (KHTML, like Gecko) Version/17.0 Mobile/15E148 Safari/604.1",
384 "Mozilla/5.0 (iPad; CPU OS 15_5 like Mac OS X) AppleWebKit/605.1.15 (KHTML, like Gecko) Version/15.0 Mobile/15E148 Safari/604.1",
385
386
387 "Mozilla/5.0 (iPhone; CPU iPhone OS 17_0 like Mac OS X) AppleWebKit/537.36 (KHTML, like Gecko) CriOS/123.0.0.0 Mobile/15E148 Safari/604.1",
388
389
390 "Mozilla/5.0 (Linux; Android 13; Pixel 6) AppleWebKit/537.36 (KHTML, like Gecko) Chrome/122.0.0.0 Mobile Safari/537.36 OPR/76.0.4017.123",
391
392
393 "Mozilla/5.0 (Linux; Android 13; Pixel 6 Pro) AppleWebKit/537.36 (KHTML, like Gecko) Chrome/123.0.0.0 Mobile Safari/537.36 EdgA/123.0.2420.64",
394
395
396 "Mozilla/5.0 (Android 13; Mobile; rv:124.0) Gecko/124.0 Firefox/124.0",
397 "Mozilla/5.0 (Android 12; Mobile; rv:122.0) Gecko/122.0 Firefox/122.0"
398]
399
400
401 all_user_agents = desktop_user_agents + mobile_user_agents
402 random_user_agent = random.choice(all_user_agents)
403
404
405 qry="https://google.com/search?q="+ concatstring
406 Actor.log.info(concatstring)
407
408 gl_map = {
409 "it": "it",
410 "www": "us",
411 "gm": "de",
412 "fr": "fr",
413 "sp": "es",
414 "uk": "uk",
415 "al": "al",
416 "ag": "dz",
417 "ar": "ar",
418 "am": "am",
419 "as": "au",
420 "au": "at",
421 "aj": "az",
422 "bg": "bd",
423 "bo": "by",
424 "be": "be",
425 "bh": "bz",
426 "bk": "ba",
427 "br": "br",
428 "bu": "bg",
429 "ca": "ca",
430 "ci": "cl",
431 "ch": "cn",
432 "co": "co",
433 "cs": "cr",
434 "hr": "hr",
435 "cu": "cu",
436 "ez": "ez",
437 "dk": "dk",
438 "ec": "ec",
439 "eg": "eg",
440 "en": "ee",
441 "fi": "fi",
442 "gg": "ge",
443 "gr": "gr",
444 "hk": "hk",
445 "hu": "hu",
446 "ic": "is",
447 "in": "in",
448 "id": "id",
449 "ir": "ir",
450 "iz": "iq",
451 "ei": "ie",
452 "is": "il",
453 "jm": "jm",
454 "ja": "jp",
455
456 "ke": "ke",
457 "kn": "kp",
458 "ks": "kr",
459 "ku": "kw",
460 "lg": "lv",
461 "ly": "ly",
462 "ls": "li",
463 "lh": "lt",
464 "lu": "lu",
465 "mc": "mo",
466 "mk": "mk",
467 "my": "my",
468 "mt": "mt",
469 "mx": "mx",
470 "md": "md",
471 "mn": "mc",
472 "mj": "me",
473 "mo": "ma",
474 "np": "np",
475 "nl": "nl",
476 "nz": "nz",
477 "ni": "ng",
478 "no": "no",
479 "pk": "pk",
480 "we": "ps",
481 "pm": "pa",
482 "pa": "py",
483 "pe": "pe",
484 "rp": "ph",
485 "pl": "pl",
486 "po": "pt",
487 "rq": "pr",
488 "qa": "qa",
489 "ro": "ro",
490 "rs": "ru",
491 "sm": "sm",
492 "sa": "sa",
493 "sg": "sn",
494 "ri": "rs",
495 "sn": "sg",
496 "lo": "sk",
497 "si": "si",
498 "sf": "za",
499 "sw": "se",
500 "sz": "ch",
501 "sy": "sy",
502 "tw": "tw",
503 "th": "th",
504 "ts": "tn",
505 "tu": "tu",
506 "ua": "ua",
507 "ae": "ae",
508 "uy": "uy",
509 "uz": "uz",
510 "ve": "ve",
511 "vn": "vn",
512 "lk": "lk",
513 }
514
515 gl = gl_map.get(Country_val, Country_val)
516
517
518
519 all_users = []
520 query =concatstring
521 max_pages = 100
522 results = []
523 count_result=0
524 has_next=True
525 for page in range(max_pages):
526 try:
527 start = page * 10
528 Actor.log.info('Check Page '+str(page))
529 url = f"http://www.google.com/search?q={query}&num=10&hl=en&start={start}&gl={gl}"
530
531
532
533 proxies=None
534 if proxyurl:
535 proxies = {'http': proxyurl, 'https': proxyurl}
536
537 response = requests.get(
538 url,
539 proxies=proxies,
540 headers={
541 "User-Agent": "Mozilla/5.0 (Windows NT 10.0; Win64; x64) "
542 "AppleWebKit/537.36 (KHTML, like Gecko) "
543 "Chrome/120.0.0.0 Safari/537.36"
544 }
545 )
546
547 if response.status_code != 200:
548 Actor.log.warning(f"⚠️ Request failed: {response.status_code} {response.reason}")
549 continue
550
551 soup = BeautifulSoup(response.text, "html.parser")
552 result_blocks = soup.select("div.g, div.tF2Cxc")
553
554 for j, block in enumerate(result_blocks):
555 Email=''
556 title_el = block.select_one("h3")
557 link_el = block.select_one("a")
558 snippet_el = block.select_one(".VwiC3b, .IsZvec, .aCOpRe")
559
560
561
562 title= title_el.get_text(strip=True)
563 url =link_el.get("href")
564 snippet= snippet_el.get_text(strip=True)
565 print(title)
566 alltext = title+snippet
567 match = re.findall(r'[a-zA-Z0-9\.\-+_]+@[a-zA-Z0-9\.\-+_]+\.[a-zA-Z]+', alltext)
568 Actor.log.info('match email '+str(len(match)))
569 if len(match)>0:
570 for i in match:
571 Email=i
572 Actor.log.info('email '+Email)
573 else:
574 match_website = re.findall(r'\\b(?:https?://|www\\.)\\S+\\b', alltext)
575 for i in match_website:
576 Website = "http://www."+i;
577 Actor.log.info('Website '+Website)
578 Email=scrape_contact_emails(Website)
579 if Email :
580 existindb=False
581
582 if len(all_users)>0:
583 for item in all_users:
584 if item['Email'] == Email :
585 existindb=True
586 break
587
588 if existindb==False:
589 all_users.append({'Email': Email});
590 await Actor.push_data({'Email': Email, 'title': title,'Description':alltext,'Detail_Link':url})
591 count_result=count_result+1
592 if(Limit_val):
593 if Limit_val!='0':
594 if(count_result>=int(Limit_val)):
595 has_next = False
596 print('Limit Exceed')
597 break
598
599 if has_next==False:
600 break
601 await asyncio.sleep(random.uniform(1.5, 3.0))
602 except:
603 break
604
605
606 print('Check in Yahoo Search Engine')
607 query=concatstring.replace("OR", "or")
608
609
610 for page in range(max_pages):
611 try:
612 if has_next==False:
613 break
614 start = page * 10
615 Actor.log.info('Check Page '+str(page))
616
617 url = "https://search.yahoo.com/search;_ylt=Awr.2lgGCQ9p1C0DekZXNyoA;_ylu=Y29sbwNncTEEcG9zAzEEdnRpZAMEc2VjA3BhZ2luYXRpb24-"
618 params = {
619 "p": query,
620 "b": start,
621 "pz": 10,
622 "bct": 0,
623 "xargs": 0
624 }
625
626
627 split_url = urlsplit(url)
628 query_str = urlencode(params)
629 full_url = urlunsplit((split_url.scheme, split_url.netloc, split_url.path, query_str, ""))
630
631 print(f"Fetching page {full_url}")
632
633
634 proxies=None
635
636
637
638 response = requests.get(
639 full_url,
640 proxies=proxies,
641 headers={
642 "User-Agent": "Mozilla/5.0 (Windows NT 10.0; Win64; x64) "
643 "AppleWebKit/537.36 (KHTML, like Gecko) "
644 "Chrome/120.0.0.0 Safari/537.36"
645 }
646 )
647
648 if response.status_code != 200:
649 Actor.log.warning(f"⚠️ Request failed: {response.status_code} {response.reason}")
650 break
651
652 soup = BeautifulSoup(response.text, "html.parser")
653 search_results = soup.select("div.dd.algo.algo-sr")
654
655 if not search_results:
656 log("⚠️ No more results found, ending.")
657 break
658
659 for r in search_results:
660 try:
661 Email=''
662 Website=''
663 Address=''
664 BusinessName=''
665 DetailsLink=''
666 Category=''
667 link_tag = r.select_one("h3.title")
668 link_tag1 = r.select_one("a")
669 if link_tag:
670 BusinessName = link_tag.get_text(strip=True)
671 DetailsLink = link_tag1.get("href")
672 Category = urlparse(DetailsLink).netloc
673
674 addr_tag = r.select_one("div.compText.aAbs")
675 if addr_tag:
676 Address = addr_tag.get_text(separator=" ").strip()
677
678 combined = r.get_text(strip=True)
679 print(combined)
680 match = re.findall(r'[a-zA-Z0-9\.\-+_]+@[a-zA-Z0-9\.\-+_]+\.[a-zA-Z]+', combined)
681 Actor.log.info('match email '+str(len(match)))
682 if len(match)>0:
683 for i in match:
684 Email=i
685 Actor.log.info('email '+Email)
686 else:
687 match_website = re.findall(r'\\b(?:https?://|www\\.)\\S+\\b', combined)
688 for i in match_website:
689 Website = "http://www."+i;
690 Actor.log.info('Website '+Website)
691 Email=scrape_contact_emails(Website)
692 if Email :
693 existindb=False
694
695 if len(all_users)>0:
696 for item in all_users:
697 if item['Email'] == Email :
698 existindb=True
699 break
700
701 if existindb==False:
702 all_users.append({'Email': Email});
703 await Actor.push_data({'Email': Email, 'title': BusinessName,'Description':combined,'Detail_Link':DetailsLink})
704 count_result=count_result+1
705 if(Limit_val):
706 if Limit_val!='0':
707 if(count_result>=int(Limit_val)):
708 has_next = False
709 print('Limit Exceed')
710 break
711
712 except Exception as ex:
713 print(f"⚠️ Parse error: {ex}")
714 continue
715
716 if has_next==False:
717 break
718 await asyncio.sleep(random.uniform(1.5, 3.0))
719 except:
720
721 break
722 await Actor.exit()
723