Python getting incomplete next page URL (BeautifulSoup, Request)

Viewed 61

I am very new to Python and Web Scraping, http://books.toscrape.com/index.html for a project but I am stuck with the pagination logic. So far i managed to get every category, the book links and the informations i needed within them but i am struggling to scrape the next page URL for every category. The first problem is that the next page URL is incomplete (but that i can manage), the second probleme is that the base URL i have to use changes for every category. Here is my code :

import requests
from bs4 import BeautifulSoup

project = []

url = 'http://books.toscrape.com'
r = requests.get(url)
soup = BeautifulSoup(r.text, "html.parser")

links = []
categories = soup.findAll("ul", class_="nav nav-list")
for category in categories:
    hrefs = category.find_all('a', href=True)
    for href in hrefs:
        links.append(href['href'])
new_links = [element.replace("catalogue", "http://books.toscrape.com/catalogue") for element in links]
del new_links[0]

page = 0
books = []


for link in new_links:
    r2 = requests.get(link).text
    book_soup = BeautifulSoup(r2, "html.parser")
    print("category: " + link)
    nextpage = True
    while nextpage:
        book_link = book_soup.find_all(class_="product_pod")
        for product in book_link:
            a = product.find('a')
            full_link = a['href'].replace("../../..", "")
            print("book: " + full_link)
            books.append("http://books.toscrape.com/catalogue" + full_link)
        if book_soup.find('li', class_='next') is None:
            nextpage = False
            page += 1
            print("end of pagination")
        else:
            next_page = book_soup.select_one('li.next>a')
            print(next_page)

The part i am struggling is with the WHILE loop in "for link in new_links". I am mostly looking for any example that can help me. Thank you!

1 Answers

If you do not want to scrape the links via http://books.toscrape.com/index.html directly while paging all the results you could get your goal like this:

from bs4 import BeautifulSoup
import requests

base_url = 'http://books.toscrape.com/'
soup = BeautifulSoup(requests.get(base_url).text)

books = []

for cat in soup.select('.nav-list ul a'):
    
    cat_url = base_url+cat.get('href').rsplit('/',1)[0]
    url = cat_url
    while True:
        soup = BeautifulSoup(requests.get(url).text)
        ##print(url)
        books.extend(['http://books.toscrape.com/catalogue/'+a.get('href').strip('../../../') for a in soup.select('article h3 a')])
        if soup.select_one('li.next a'):
            url = f"{cat_url}/{soup.select_one('li.next a').get('href')}"
        else:
            break
books

Cause the result would be the same I would recommend to skip the way over categories:

from bs4 import BeautifulSoup
import requests

baseurl = 'http://books.toscrape.com/'
url = 'https://books.toscrape.com/catalogue/page-1.html'
soup = BeautifulSoup(requests.get(base_url).text)

books = []

while True:
    soup = BeautifulSoup(requests.get(url).text)
    
    for a in soup.select('article h3 a'):
        bsoup = BeautifulSoup(requests.get(base_url+'catalogue/'+a.get('href')).content)
        print(base_url+'catalogue/'+a.get('href'))
        data = {
            'title': bsoup.h1.text.strip(),
            'category': bsoup.select('.breadcrumb li')[-2].text.strip(),
            'url': base_url+'catalogue/'+a.get('href')
            ### add what ever is needed
        }
        data.update(dict(row.stripped_strings for row in bsoup.select('table tr')))
        books.append(data)
    if soup.select_one('li.next a'):
        url = f"{url.rsplit('/',1)[0]}/{soup.select_one('li.next a').get('href')}"
    else:
        break
books
Output
[{'title': 'A Light in the Attic',
  'category': 'Poetry',
  'url': 'http://books.toscrape.com/catalogue/a-light-in-the-attic_1000/index.html',
  'UPC': 'a897fe39b1053632',
  'Product Type': 'Books',
  'Price (excl. tax)': '£51.77',
  'Price (incl. tax)': '£51.77',
  'Tax': '£0.00',
  'Availability': 'In stock (22 available)',
  'Number of reviews': '0'},
 {'title': 'Tipping the Velvet',
  'category': 'Historical Fiction',
  'url': 'http://books.toscrape.com/catalogue/tipping-the-velvet_999/index.html',
  'UPC': '90fa61229261140a',
  'Product Type': 'Books',
  'Price (excl. tax)': '£53.74',
  'Price (incl. tax)': '£53.74',
  'Tax': '£0.00',
  'Availability': 'In stock (20 available)',
  'Number of reviews': '0'},
 {'title': 'Soumission',
  'category': 'Fiction',
  'url': 'http://books.toscrape.com/catalogue/soumission_998/index.html',
  'UPC': '6957f44c3847a760',
  'Product Type': 'Books',
  'Price (excl. tax)': '£50.10',
  'Price (incl. tax)': '£50.10',
  'Tax': '£0.00',
  'Availability': 'In stock (20 available)',
  'Number of reviews': '0'},...]
Related