Indeed Job Scraper with Python - only returning url links, no title/description

Viewed 37

I know you've probably seen 100 Indeed Scraping posts on here, and i'm hoping mine is a bit different. Essentially, I'm trying to build an Indeed job scraper that pulls company name and job title, based on a search with "job title" and "location" being variables. Additionally, when selenium opens chrome, Indeed is auto-populating my location, which doesn't get overwritten by the location I've inputting in the code.

I'm fairly new to Python, and I'm relying on the foundation built by someone else from GitHub, so I am having trouble diagnosing the problem.

Would love any help or insight!

Here is my code:

from selenium.webdriver.common.keys import Keys
from selenium.webdriver.support.ui import WebDriverWait
from selenium.webdriver.support import expected_conditions as EC
from selenium.webdriver.common.by import By
from selenium.common.exceptions import TimeoutException
import json
from time import sleep    

list_of_description = ["warehouse","associate"]
URL = "https://www.indeed.com/"
MAIN_WINDOW_HANDLER = 0
JOB_TITLE = " "
JOB_LOCATION = " "
JSON_DICT_ARRAY = []

def main():
    pageCounter = 0
    bool_next = True
    newUrl = ""
    # theUrl = "https://ca.indeed.com/jobs?q=developer&l=Winnipeg%2C+MB"

    browser = webdriver.Chrome(service=Service(ChromeDriverManager().install()))
    browser.get( URL )

    # Change text in where
    whatElement = browser.find_element(By.ID,"text-input-what")
    whatElement.send_keys( JOB_TITLE )

    # Change text in where
    whereElement = browser.find_element(By.ID,"text-input-where")
    whereElement.send_keys(Keys.CONTROL + "a")
    whereElement.send_keys(Keys.BACK_SPACE)
    whereElement.send_keys( JOB_LOCATION )
    whereElement.submit()

    MAIN_WINDOW_HANDLER = browser.window_handles[0]

    fileName = "{} Jobs in {}.json".format(JOB_TITLE, JOB_LOCATION)

    newPage = True
    nextNumber = 2
    searchPhrase = '//span[contains(text(), "{0}") and @class="pn"]'.format(nextNumber)
    currentHTML = browser.page_source
    linkElements = getElementFromHTML('div .title', currentHTML) # searching for div tags with title class
    reqResultText = currentHTML #(download_file(URL)).text
    browser.get( browser.current_url )
    browser.get( browser.current_url )
    scrapeJobListing(linkElements, reqResultText, browser, MAIN_WINDOW_HANDLER)

    if( check_exists_by_xpath(browser, '//button[@id="onetrust-accept-btn-handler"]') ):
        try:
            theElement = browser.find_element(By.XPATH, '//button[@id="onetrust-accept-btn-handler"]' )
            print(type(theElement))
            theElement.click()
            print("I clicked")
            # scrapeJobListing(linkElements, reqResultText, browser, MAIN_WINDOW_HANDLER)
            while ( newPage and check_exists_by_xpath(browser, searchPhrase) ):
                theElement = browser.find_elements(By.XPATH, searchPhrase )
                try:
                    theElement[0].click()
                except:
                    newPage = False
                if(newPage):
                    browser.get(browser.current_url)
                    print(browser.current_url)
                    nextNumber += 1
                    searchPhrase = '//span[contains(text(), "{0}") and @class="pn"]'.format(nextNumber)
                    currentHTML = browser.page_source
                    linkElements = getElementFromHTML('div .title', currentHTML) # searching for div tags with title class
                    reqResultText = currentHTML #(download_file(URL)).text
                    scrapeJobListing(linkElements, reqResultText, browser, MAIN_WINDOW_HANDLER)
                
                else:
                    print ("Search Concluded")
        except:
            # scrapeJobListing(linkElements, reqResultText, browser, MAIN_WINDOW_HANDLER)
            while ( newPage and check_exists_by_xpath(browser, searchPhrase) ):
                theElement = browser.find_elements(By.XPATH, searchPhrase )
                try:
                    theElement[0].click()
                except:
                    newPage = False
                if(newPage):
                    browser.get(browser.current_url)
                    print(browser.current_url)
                    nextNumber += 1
                    searchPhrase = '//span[contains(text(), "{0}") and @class="pn"]'.format(nextNumber)
                    currentHTML = browser.page_source
                    linkElements = getElementFromHTML('div .title', currentHTML) # searching for div tags with title class
                    reqResultText = currentHTML #(download_file(URL)).text
                    scrapeJobListing(linkElements, reqResultText, browser, MAIN_WINDOW_HANDLER)
                else:
                    print ("Search Concluded")
    
    with open(fileName, "w") as data:
        for it in JSON_DICT_ARRAY:
            data.write(json.dumps(it))
            data.write(",\n")
        data.close()

def scrapeJobListing(linkElements, reqResultText, browser, mainHandler):
    jobDes = ""
    for i in range( len(linkElements) ):
        print("\n ",i)
        jsonDataDict = {}
        list = re.findall(r'["](.*?)["]',str(linkElements[i]))
        currJobMap = "jobmap[{}]= ".format(i)
        openBracketIndex = reqResultText.find(currJobMap) + len(currJobMap)
        findNewString = reqResultText[openBracketIndex:openBracketIndex+600]
        print (findNewString)
        
        closeBracketIndex = findNewString.find("}") + 1
        cmpOpen = findNewString.find("cmp:'") + len("cmp:'")
        cmpClose = findNewString.find("',cmpesc:")
        titleOpen = findNewString.find("title:'") + len("title:'")
        titleClose = findNewString.find("',locid:")
        parsedString = str( findNewString[0:closeBracketIndex] )
        print (parsedString)
        print("\n")
        cmpName = parsedString[cmpOpen:cmpClose]# Company Name
        jobTitle = parsedString[titleOpen:titleClose]# Job Title
        jsonDataDict['(2) Company Name'] = cmpName
        jsonDataDict['(1) Job Title'] = jobTitle

        try:
            title = browser.find_element(By.ID,list[4]) # 4th quotation is the Job Description

            print('Found <%s> element with that class name!' % (title.tag_name))
            title.click()
            window_after = browser.window_handles[1]
            browser.switch_to.window(window_after)
            theCurrURL = browser.current_url
            browser.get(theCurrURL)
            currPageSource = browser.page_source
            jsonDataDict['(4) Job Link'] = theCurrURL

            print (theCurrURL)
            jobDes = getElementFromHTML('div #jobDescriptionText', currPageSource)
            soup = bs4.BeautifulSoup(str(jobDes), "html.parser")
            jobDescText = soup.get_text('\n')
            jsonDataDict['(3) Job Description'] = jobDescText
            JSON_DICT_ARRAY.append(jsonDataDict)
            browser.close()
            print(jobDes)
        except:
            print('Was not able to find an element with that name.')
        
        # sleep(2)

    print (mainHandler)
    browser.switch_to.window(mainHandler) #Not necessary right?

def getElementBySearch(searchTag, theURL):
    reqResult = download_file(theURL)
    soup = bs4.BeautifulSoup(reqResult.text, "html.parser")
    element = soup.select(searchTag)

    return element

def getElementFromHTML(searchTag, htmlText):
    soup = bs4.BeautifulSoup(htmlText, "html.parser")
    element = soup.select(searchTag)

    return element 

def check_exists_by_xpath(webdriver, xpath):
    try:
        webdriver.find_elements(By.XPATH,xpath)
    except NoSuchElementException:
        return False
    return True

def download_file(searchPhrase):
    result = requests.get(searchPhrase)
    # type(result)

    # Check for error
    try:
        result.raise_for_status()
    except Exception as exc:
        print('There was a problem: %s' % (exc))

    return result

if __name__== "__main__":
    main()

Right now, the script essentially opens Indeed, looks through each page and posts the links. But I'm not sure why it's not providing the job title and company information.

The output looks like this -

https://www.indeed.com/jobs?q=Warehouse&l=Eugene,%20Oregon&radius=50&start=10&vjk=cbe41d08db5e3eaa
CDwindow-07A1095B627FCB670750FBBDC552B60D
https://www.indeed.com/jobs?q=Warehouse&l=Eugene,%20Oregon&radius=50&start=20&vjk=cbe41d08db5e3eaa
CDwindow-07A1095B627FCB670750FBBDC552B60D
https://www.indeed.com/jobs?q=Warehouse&l=Eugene,%20Oregon&radius=50&start=30&vjk=cbe41d08db5e3eaa
CDwindow-07A1095B627FCB670750FBBDC552B60D
https://www.indeed.com/jobs?q=Warehouse&l=Eugene,%20Oregon&radius=50&start=40&vjk=2fd38d5eb42b6ca4
CDwindow-07A1095B627FCB670750FBBDC552B60D
https://www.indeed.com/jobs?q=Warehouse&l=Eugene,%20Oregon&radius=50&start=50&vjk=acbe5c6e5afce1d6
CDwindow-07A1095B627FCB670750FBBDC552B60D
https://www.indeed.com/jobs?q=Warehouse&l=Eugene,%20Oregon&radius=50&start=60&vjk=acbe5c6e5afce1d6
CDwindow-07A1095B627FCB670750FBBDC552B60D
https://www.indeed.com/jobs?q=Warehouse&l=Eugene,%20Oregon&radius=50&start=70&vjk=acbe5c6e5afce1d6
CDwindow-07A1095B627FCB670750FBBDC552B60D
https://www.indeed.com/jobs?q=Warehouse&l=Eugene,%20Oregon&radius=50&start=80&vjk=acbe5c6e5afce1d6
CDwindow-07A1095B627FCB670750FBBDC552B60D
Search Concluded
0 Answers
Related