Here's my script :
To bypass datadome, I used it not at once, first :
import pandas as pd
import numpy as np
import time
import random
from selenium import webdriver
from selenium.webdriver.support.select import Select
from selenium.webdriver.support.ui import WebDriverWait
from selenium.webdriver.common.by import By
from selenium.webdriver.support import expected_conditions as EC
from selenium.webdriver.common.keys import Keys
PATH = "chromedriver.exe"
options = webdriver.ChromeOptions()
options.add_argument("--disable-gpu")
options.add_argument('enable-logging')
options.add_argument("start-maximized")
options.add_experimental_option("excludeSwitches", ["enable-automation"])
options.add_experimental_option('useAutomationExtension', False)
driver = webdriver.Chrome(options=options, executable_path=PATH)
driver.execute_script("Object.defineProperty(navigator, 'webdriver', {get: () => undefined})")
driver.execute_cdp_cmd('Network.setUserAgentOverride', {"userAgent": 'Mozilla/5.0 (Windows NT 10.0; Win64; x64) AppleWebKit/537.36 (KHTML, like Gecko) Chrome/83.0.4103.53 Safari/537.36'})
url = 'https://www.leboncoin.fr/voitures/offres'
driver.get(url)
Then I make manually the protection, then second part :
cookie = driver.find_element_by_xpath('//*[@id="didomi-notice-agree-button"]')
try:
cookie.click()
except:
pass
time.sleep(2)
car = driver.find_element_by_xpath('//input[@autocomplete="search-keyword-suggestions"]')
car.click()
car.send_keys('Peugeot')
car.send_keys(Keys.ENTER)
time.sleep(3)
for x in range(10):
#time.sleep(3)
links = driver.find_elements_by_xpath('//span[@class="_137P- P4PEa _35DXM"]')
for l in links:
data = l.text
#x = data.split("/n")
print(data)
links2 = driver.find_elements_by_xpath('//p[@data-qa-id="aditem_title"]')
for i in links2:
data2 = i.text
#x = data.split("/n")
print(data2)
print(data2)
print(data2)
print(data2)
print(data2)
links3 = driver.find_elements_by_xpath('//a[@class="sc-jlyJG loRlDv"]')
for j in links3:
data3 = j.get_attribute("href")
#x = data.split("/n")
print(data3)
print(data3)
print(data3)
print(data3)
print(data3)
links4 = driver.find_elements_by_xpath('//span[@class="_2k43C Dqdzf cJtdT _3j0OU"]')
for k in links4:
data4 = k.text
#x = data.split("/n")
print(data4)
print(data4)
print(data4)
print(data4)
print(data4)
next = driver.find_element_by_xpath('//a[@title="Page suivante"]')
next.click()
time.sleep(3)
But, as you can see, my code isn't optimal. I obtained something like that :
7 490 €
2015
52000 km
Essence
Manuelle
...
Peugeot 407 hdi 2l exécutive 136cv
Peugeot 407 hdi 2l exécutive 136cv
Peugeot 407 hdi 2l exécutive 136cv
Peugeot 407 hdi 2l exécutive 136cv
Peugeot 407 hdi 2l exécutive 136cv
...
https://www.leboncoin.fr/voitures/2022493158.htm?ac=4051296304
https://www.leboncoin.fr/voitures/2022493158.htm?ac=4051296304
https://www.leboncoin.fr/voitures/2022493158.htm?ac=4051296304
https://www.leboncoin.fr/voitures/2022493158.htm?ac=4051296304
https://www.leboncoin.fr/voitures/2022493158.htm?ac=4051296304
...
Saint-Pardoux-Corbier 19210
Saint-Pardoux-Corbier 19210
Saint-Pardoux-Corbier 19210
Saint-Pardoux-Corbier 19210
Saint-Pardoux-Corbier 19210
I need to c/c manually into a csv file to obtained that :
But that not convenient at all. If you have a way to obtain this csv immediately, that would be awesome.
