Unable to scrape emails from some websites maybe due to r.html.render() not working properly

Viewed 62

I have some website links as samples for extracting any email available in their internal sites.

However, even I am trying to render any JS driven website via r.html.render() within scrape_email(url) method, some of the websites like arken.trygge.dk, gronnebakken.dk, dagtilbud.ballerup.dk/boernehuset-bispevangen etc. does not return any email which might be due to rendering issue.

I have attached the sample file for convenience of running

I dont want to use selenium as there can be thousands or millions of webpage I want to extract emails from.

So far this is my code:

import os
import time
import requests
from urllib.parse import urlparse, urljoin
from bs4 import BeautifulSoup
import re
from requests_html import HTMLSession
import pandas as pd
from gtts import gTTS
import winsound
# For convenience of seeing console output in the script
pd.options.display.max_colwidth = 180

#Get the start time of script execution
startTime = time.time()

#Paste file name inside ''
input_file_name = 'sample'

input_df = pd.read_excel(input_file_name+'.xlsx', engine='openpyxl')
input_df = input_df.dropna(how='all')

internal_urls = set()
emails = set()
total_urls_visited = 0


def is_valid(url):
    """
    Checks whether `url` is a valid URL.
    """
    parsed = urlparse(url)
    return bool(parsed.netloc) and bool(parsed.scheme)


def get_internal_links(url):
    """
    Returns all URLs that is found on `url` in which it belongs to the same website
    """
    # all URLs of `url`
    urls = set()
    # domain name of the URL without the protocol
    domain_name = urlparse(url).netloc
    print("Domain name -- ",domain_name)
    try:       
        soup = BeautifulSoup(requests.get(url, timeout=5).content, "html.parser")
        for a_tag in soup.findAll("a"):
            href = a_tag.attrs.get("href")
            if href == "" or href is None:
                # href empty tag
                continue
            # join the URL if it's relative (not absolute link)
            href = urljoin(url, href)
            parsed_href = urlparse(href)
            # remove URL GET parameters, URL fragments, etc.
            href = parsed_href.scheme + "://" + parsed_href.netloc + parsed_href.path
            if not is_valid(href):
                # not a valid URL
                continue
            if href in internal_urls:
                # already in the set
                continue
            if parsed_href.netloc != domain_name:
                # if the link is not of same domain pass
                continue
            if parsed_href.path.endswith((".csv",".xlsx",".txt", ".pdf", ".mp3", ".png", ".jpg", ".jpeg", ".svg", ".mov", ".js",".gif",".mp4",".avi",".flv",".wav")):
                # Overlook site images,pdf and other file rather than webpages
                continue

            print(f"Internal link: {href}")
            urls.add(href)
            internal_urls.add(href)
        return urls
    
    except requests.exceptions.Timeout as err:
        print("The website is not loading within 5 seconds... Continuing crawling the next one")
        pass
    except:
        print("The website is unavailable. Continuing crawling the next one")
        pass


def crawl(url, max_urls=30):
    """
    Crawls a web page and extracts all links.
    You'll find all links in `external_urls` and `internal_urls` global set variables.
    params:
        max_urls (int): number of max urls to crawl, default is 30.
    """
    global total_urls_visited
    total_urls_visited += 1
    print(f"Crawling: {url}")
    links = get_internal_links(url)
    # for link in links:
    #     if total_urls_visited > max_urls:
    #         break
        # crawl(link, max_urls=max_urls)


def scrape_email(url):
    EMAIL_REGEX = r'\b[A-Za-z0-9._%+-]+@[A-Za-z0-9.-]+\.[A-Z|a-z]{2,}\b'
    # EMAIL_REGEX = r"""(?:[a-z0-9!#$%&'*+/=?^_`{|}~-]+(?:\.[a-z0-9!#$%&'*+/=?^_`{|}~-]+)*|"(?:[\x01-\x08\x0b\x0c\x0e-\x1f\x21\x23-\x5b\x5d-\x7f]|\\[\x01-\x09\x0b\x0c\x0e-\x7f])*")@(?:(?:[a-z0-9](?:[a-z0-9-]*[a-z0-9])?\.)+[a-z0-9](?:[a-z0-9-]*[a-z0-9])?|\[(?:(?:(2(5[0-5]|[0-4][0-9])|1[0-9][0-9]|[1-9]?[0-9]))\.){3}(?:(2(5[0-5]|[0-4][0-9])|1[0-9][0-9]|[1-9]?[0-9])|[a-z0-9-]*[a-z0-9]:(?:[\x01-\x08\x0b\x0c\x0e-\x1f\x21-\x5a\x53-\x7f]|\\[\x01-\x09\x0b\x0c\x0e-\x7f])+)\])"""
    try:
        # initiate an HTTP session
        session = HTMLSession()
        # get the HTTP Response
        r = session.get(url, timeout=10)
        # for JAVA-Script driven websites
        r.html.render()
        single_url_email = []
        for re_match in re.finditer(EMAIL_REGEX, r.html.raw_html.decode()):
            single_url_email.append(re_match.group().lower())
        r.session.close()
        return set(single_url_email)
    except:
        pass

def crawl_website_scrape_email(url, max_internal_url_no=20):
    crawl(url,max_urls=max_internal_url_no)
    each_url_emails = []
    global internal_urls
    global emails
    for each_url in internal_urls:
        each_url_emails.append(scrape_email(each_url))
        
    URL_WITH_EMAILS={'main_url': url, 'emails':each_url_emails}
    emails = {}
    internal_urls = set()
    return URL_WITH_EMAILS


def list_check(emails_list, email_match):
    match_indexes = [i for i, s in enumerate(emails_list) if email_match in s]
    return [emails_list[index] for index in match_indexes]

URL_WITH_EMAILS_LIST = [crawl_website_scrape_email(x) for x in input_df['Website'].values]
URL_WITH_EMAILS_DF = pd.DataFrame(data = URL_WITH_EMAILS_LIST)
URL_WITH_EMAILS_DF.to_excel(f"{input_file_name}_email-output.xlsx", index=False)

How can I solve the issue of not being able to scrape email from some of those above-mentioned and similar type of websites?

Is there also any way to detect and print strings if my get request is refused by bot detector or related protocols?

Also how can I make this code more robust?

Thank you in advance

0 Answers
Related