This code scrolls down to the bottom of the given link in the code, and should append all the URLs of the recipes' pages given in there. But it only URLs randomly up-to 70/80/90. I don't understand why that is happening.
import time
from bs4 import BeautifulSoup
import requests
from selenium import webdriver
from selenium.webdriver.common.keys import Keys
import numpy as np
from urllib.parse import urljoin
driver = webdriver.Firefox()
driver.get("https://www.kitchenstories.com/en/categories/vegan-dishes")
time.sleep(2)
scroll_pause_time = 0.1
screen_height = driver.execute_script("return window.screen.height;")
i = 1
screen_height = screen_height/16
while True:
driver.execute_script("window.scrollTo(0, {screen_height}*{i});".format(screen_height=screen_height, i=i))
i += 1
time.sleep(scroll_pause_time)
scroll_height = driver.execute_script("return document.body.scrollHeight;")
print(scroll_height)
print(screen_height)
if (screen_height) * i > scroll_height*2:
break
urls = []
soup = BeautifulSoup(driver.page_source, "html.parser")
for parent in soup.find_all('li',class_ ="cursor-pointer col-span-6 col-start-auto w-full sm:col-span-4 md:col-span-3"):
a_tag = parent.find('li',class_ ="cursor-pointer col-span-6 col-start-auto w-full sm:col-span-4 md:col-span-3")
base = "https://www.kitchenstories.com/en/categories/vegan-dishes"
link = parent.a['href']
url = urljoin(base,link)
urls.append(url)
print(len(url))
print(urls)
what I am expecting out of this code is for it to give me a list of all the 522 URLs so that I can further scrape off the ingredients for those respective pages in the website.