add_done_call_back() gives AttributeError: 'Future' object has no attribute 'select'

Viewed 3042

I tried making a simple web crawler.

I want to create a data frame with the parent link and the links that are found on the parent link page along with the text associated with it.

enter image description here

import requests
from bs4 import BeautifulSoup
from queue import Queue, Empty
from concurrent.futures import ThreadPoolExecutor
from urllib.parse import urljoin
import urllib
from urllib.error import HTTPError #used in the main function, to catch HTTPErrors
from requests.exceptions import InvalidURL #used in the main function, to catch invalidUrl errors
from urllib.parse import urlparse #used to parse the homepage url and get network location out of it
from urllib.parse import quote #used to correct incorrect urls
import pandas as pd


class MultiThreadScraper:

    global df
    df = pd.DataFrame(data=None, columns = ['parent_link','link', 'text'])
    
    def __init__(self, base_url):

        self.base_url = base_url
        self.root_url = '{}://{}'.format(urlparse(self.base_url).scheme, urlparse(self.base_url).netloc)
        self.pool = ThreadPoolExecutor(max_workers=20)
        self.scraped_pages = set([])
        self.to_crawl = Queue()
        self.to_crawl.put(self.base_url)
    
    #gets http.client.HTTPResponse from the server
    def get_http_response(self, url):
        header = {'User-Agent': 'Mozilla/5.0 (X11; Linux x86_64) AppleWebKit/537.11 (KHTML, like Gecko) Chrome/87.0.4280.88 Safari/537.11',
                      'Accept': 'text/html,application/xhtml+xml,application/xml;q=0.9,*/*;q=0.8',
                      'Accept-Charset': 'ISO-8859-1,utf-8;q=0.7,*;q=0.3',
                      'Accept-Encoding': 'none',
                      'Accept-Language': 'en-US,en;q=0.8',
                      'Connection': 'keep-alive'}
        request = urllib.request.Request(url, headers = header) #urllib.request.Request object
        response = urllib.request.urlopen(request)  #http.client.HTTPResponse
        return response

    #accepts the http.client.HTTPResponse from the server and fetches the html content using BeautifulSoup
    def get_html_content(self, httpResponse):
        httpResponse_content = httpResponse.read() #data in bytes
        httpResponse_htmlContent = BeautifulSoup(httpResponse_content, 'html.parser')
        return httpResponse_htmlContent
        
    def parse_links(self, html,df):
        all_tags_with_hrefs = html.select('[href]')
        for link in all_tags_with_hrefs:
            url = link['href']
            if url.startswith('/') or url.startswith(self.root_url):
                url = urljoin(self.root_url, url)
                if (link.text).strip()!='':
                    df.loc[-1,'parent_link']= self.base_url
                    df.loc[-1,'link'] = url
                    df.loc[-1,'text'] = (link.text).strip() 
                    df.drop_duplicates(subset='link', inplace=True)
                    df.reset_index(drop=True, inplace=True)
                if url not in self.scraped_pages:
                    self.to_crawl.put(url)

    def post_scrape_callback(self, html_content):
            self.parse_links(html_content, df)

    def scrape_page(self, url):
        try:
            url = urllib.parse.quote(url, safe='/,:,-,?,=,&')
            get_response = self.get_http_response(url)
            try: 
                html_content = self.get_html_content(get_response)
            except:
                print(f'{url} : ERROR READING RESPONSE !')
        except(HTTPError, InvalidURL):
            print(f'{url} : NO RESPONSE !')
        return html_content
        

    def run_scraper(self):
        while True:
            try:
                target_url = self.to_crawl.get(timeout=10)
                if target_url not in self.scraped_pages:
                    print("Scraping URL: {}".format(target_url))
                    self.scraped_pages.add(target_url)
                    job = self.pool.submit(self.scrape_page, target_url)
                    job.add_done_callback(self.post_scrape_callback)
            except Empty:
                return
            except Exception as e:
                print(e)
                continue
if __name__ == '__main__':
    s = MultiThreadScraper("https://www.nationalgrid.com/")
    s.run_scraper() 

It gives me the following error : AttributeError: 'Future' object has no attribute 'select' From documentation I understood that add_done_callback() adds a callback to be run when the Future is done. Any help would be appreciated!

1 Answers

You will need a pool and to split your code logic to smaller functions that can run independently. For example if you write a crawl function that receives a URL as the argument you can do something like:

import concurrent.futures
urls = [] # a url list you create
with concurrent.futures.ThreadPoolExecutor(max_workers=5) as executor:
    for url in urls:
        executor.submit(crawl, url)

Then you can parse multiple URLs using multiple threads (here it's limited to 5)

Related